simframe 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/actions.js CHANGED
@@ -6,6 +6,12 @@ import * as api from './index.js';
6
6
  import * as graph from './graph.js';
7
7
  import * as input from './input.js';
8
8
  import * as intent from './intent.js';
9
+ import * as matching from './matching.js';
10
+ import * as vocabulary from './vocabulary.js';
11
+ import * as wrote from './wrote.js';
12
+ import * as supervisor from './supervisor.js';
13
+ import * as view from './view.js';
14
+ import * as planner from './planner.js';
9
15
  import * as metrics from './metrics.js';
10
16
  import * as screenmap from './screenmap.js';
11
17
  import { launchApp, openUrl, setPermission, terminateApp } from './platform/index.js';
@@ -35,6 +41,19 @@ const MAX_PAUSE_MS = 5000;
35
41
  * transitions that never happened. See docs/BENCHMARKS.md, Phase 11.
36
42
  */
37
43
  const FOCUS_STABLE_MS = 250;
44
+ /** How far a control's centre may move between the read and the readback. */
45
+ const FIELD_READBACK_RADIUS = 40;
46
+ // The OCR pass needs a far wider one, and 40 is why measuring mattered: OCR
47
+ // reports a value at the *value's* centre, not the label's tap point, and on a
48
+ // pinch-zoomed page those were 133pt apart on a field that had filled
49
+ // correctly. Generous enough for that, still short of "anywhere on screen",
50
+ // because a match found anywhere would happily confirm text that was already
51
+ // there before the step ran.
52
+ const FIELD_READBACK_OCR_RADIUS = 220;
53
+ // How long the OCR readback waits for a frame that postdates the keystrokes.
54
+ // Focusing a web input makes Safari re-zoom, so the settle here is doing real
55
+ // work rather than padding.
56
+ const READBACK_SETTLE_MS = 900;
38
57
  /**
39
58
  * The cold defaults, unchanged, for a field this screen has not been measured
40
59
  * focusing. `graph.focusPlan` takes over once it has been, and may only make
@@ -46,6 +65,16 @@ const FOCUS_STABLE_MS = 250;
46
65
  */
47
66
  const FOCUS_REACTION_MS = 900;
48
67
  const FOCUS_TIMEOUT_MS = 3000;
68
+
69
+ /**
70
+ * The settle budget for an action that is not supposed to navigate.
71
+ *
72
+ * A change-based settle cannot be satisfied by an action that changes nothing,
73
+ * so the only question is how long to spend finding that out. Long enough for
74
+ * keystrokes to land and no longer.
75
+ */
76
+ const STAYS_PUT_STILLNESS_MS = 200;
77
+ const STAYS_PUT_BUDGET_MS = 600;
49
78
  const POLL_MS = 250;
50
79
  /** A list that has not produced the target in this many screens does not contain it. */
51
80
  const MAX_SCROLLS = 20;
@@ -57,7 +86,7 @@ const MAX_SCROLLS = 20;
57
86
  * permission can change what the app shows.
58
87
  */
59
88
  const ACTION_STEPS = new Set([
60
- 'tap', 'tapAt', 'type', 'paste', 'swipe', 'scroll', 'scrollTo', 'button', 'key',
89
+ 'tap', 'tapAt', 'type', 'paste', 'clear', 'swipe', 'scroll', 'scrollTo', 'button', 'key',
61
90
  'launch', 'terminate', 'openUrl', 'confirm', 'chooseAny', 'permission',
62
91
  ]);
63
92
 
@@ -90,6 +119,16 @@ export function wrongTurnFrom(verification) {
90
119
  return verification?.verdict === 'unexpected-screen';
91
120
  }
92
121
 
122
+ /**
123
+ * What a step was asked to act on, in the caller's words.
124
+ *
125
+ * Every escalation wants this and two sites were spelling it out separately —
126
+ * one of them not at all.
127
+ */
128
+ export function goalOf(step = {}) {
129
+ return step.value ?? step.target ?? step.label ?? step.into ?? null;
130
+ }
131
+
93
132
  /**
94
133
  * What a halted step does to the run as a whole.
95
134
  *
@@ -131,6 +170,19 @@ export async function runScript(
131
170
  // compared against a human baseline.
132
171
  flowName = null,
133
172
  minSteps = null,
173
+ // What the plan wants its supervisor to know.
174
+ //
175
+ // The owner's insight, and it is what made the supervisor usable. Asked
176
+ // cold, it called a list that was plainly still arriving a dead end —
177
+ // because it does not know this app and Claude, by the time it writes the
178
+ // plan, does. So the plan briefs its own first responder: `supervise` is
179
+ // the batch's standing guidance, and a step may add `expect` of its own.
180
+ //
181
+ // It is also the correction channel. When the supervisor stops a run, the
182
+ // result says so, names the remaining steps, and tells the caller that
183
+ // re-issuing them with a `supervise` note will not stop again for that
184
+ // reason. A wrong local decision costs one message, not a re-plan.
185
+ supervise = null,
134
186
  options,
135
187
  } = {},
136
188
  ) {
@@ -143,6 +195,7 @@ export async function runScript(
143
195
  const flowId = metrics.newFlowId();
144
196
  const escalations = [];
145
197
  const verdicts = [];
198
+ const supervisions = [];
146
199
  // Named noteEscalation, not note: the step loop below declares its own
147
200
  // `note` string for the no-visible-change suffix, which shadowed this and
148
201
  // turned every escalating verdict into a failed step reading "note is not a
@@ -204,6 +257,8 @@ export async function runScript(
204
257
  }
205
258
  };
206
259
 
260
+ let consecutiveFailures = 0;
261
+ let lastFailureScreen = null;
207
262
  for (const [i, raw] of steps.entries()) {
208
263
  const step = normalizeStep(raw);
209
264
  const stepStart = Date.now();
@@ -243,7 +298,105 @@ export async function runScript(
243
298
  }),
244
299
  observedMs: null,
245
300
  };
246
- let detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
301
+ let detail;
302
+ // A selector that did not resolve gets the step's own alternatives before
303
+ // the batch is abandoned. Anything else propagates: retrying from a screen
304
+ // we did not expect to be on is not a retry, it is a second guess.
305
+ try {
306
+ detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
307
+ } catch (thrown) {
308
+ let err = thrown;
309
+ // Ask the supervisor before anything is abandoned. It sits behind the
310
+ // hands and in front of the reasoner: first responder, not
311
+ // decision-maker, and its whole vocabulary is wait/retry/stop.
312
+ const ruling = await superviseFailure(deviceQuery, {
313
+ goal: supervise ?? flowName, step, expected: step.expect, err, options,
314
+ });
315
+ if (ruling?.decision === 'wait' || ruling?.decision === 'retry') {
316
+ // Both wait and retry settle first, differing only in how long.
317
+ //
318
+ // The one distinction the model still fumbles is wait against retry —
319
+ // it called a loading list `retry` on the bench, which without a
320
+ // settle would fail again immediately for the same reason. Making the
321
+ // difference a duration rather than a behaviour means a wrong choice
322
+ // between them costs milliseconds instead of the recovery. The
323
+ // decision that actually matters is stop against continue, and on
324
+ // that it has been right every time it was asked something code could
325
+ // not already answer.
326
+ await api.waitFor(deviceQuery, {
327
+ mode: 'settle',
328
+ stableMs: 400,
329
+ timeoutMs: ruling.decision === 'wait' ? SUPERVISOR_WAIT_MS : SUPERVISOR_RETRY_MS,
330
+ options,
331
+ }).catch(() => null);
332
+ try {
333
+ detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
334
+ detail += ` [the local supervisor said ${ruling.decision}; it worked on the second attempt]`;
335
+ supervisions.push({ index: i, decision: ruling.decision, reason: ruling.reason, outcome: 'recovered' });
336
+ continue;
337
+ } catch (again) {
338
+ supervisions.push({ index: i, decision: ruling.decision, reason: ruling.reason, outcome: 'still failed' });
339
+ err = again;
340
+ }
341
+ } else if (!ruling && supervisor.requested(options)) {
342
+ // Attempted and got nothing. This was invisible for a whole round:
343
+ // three rulings, then twenty supervised calls and six failures with
344
+ // no output at all, while `doctor` in a separate process reported the
345
+ // model healthy. The reporter's own words — *"had I run flow 2 alone
346
+ // I would have reported the supervisor makes no difference without
347
+ // realising it had never run"*.
348
+ supervisions.push({ index: i, decision: 'unavailable', reason: 'the supervisor did not answer', outcome: 'no ruling' });
349
+ err.message += ' — the local supervisor was consulted and did not answer, so this failure was not judged.';
350
+ } else if (ruling?.decision === 'stop') {
351
+ supervisions.push({ index: i, decision: 'stop', reason: ruling.reason, from: ruling.from, outcome: 'stopped the run' });
352
+ const remaining = steps.slice(i);
353
+ err.message += ` — the local supervisor stopped the run here.`
354
+ + ` ${remaining.length} step(s) were not attempted.`
355
+ + ' If that judgement was wrong, re-issue the remaining steps with a `supervise` note'
356
+ + ' telling it what to expect, and it will not stop for this reason again.';
357
+ err.remainingSteps = remaining;
358
+ }
359
+ const { allowed, refused } = permittedAlternatives(step);
360
+ if (!allowed.length || !mayRetryAfter(err)) {
361
+ if (refused.length) {
362
+ err.message += ` (${refused.length} alternative(s) refused locally: `
363
+ + `${refused.map((r) => `"${r.label}" — ${r.reason}`).join('; ')})`;
364
+ }
365
+ throw err;
366
+ }
367
+ const tried = [step.value ?? step.target ?? step.label ?? step.into];
368
+ let last = err;
369
+ for (const label of allowed) {
370
+ try {
371
+ detail = await runStep(deviceQuery, udid, stepWithTarget(step, label), { screen, options, frames, focus });
372
+ detail += ` [after ${tried.map((t) => JSON.stringify(String(t))).join(', ')} did not resolve]`;
373
+ last = null;
374
+ break;
375
+ } catch (again) {
376
+ tried.push(label);
377
+ last = again;
378
+ if (!mayRetryAfter(again)) break;
379
+ }
380
+ }
381
+ if (last) {
382
+ // When every selector misses and the screen moved a moment ago, the
383
+ // problem is timing and not naming. Reported: three fallbacks all
384
+ // failed because an asset list had not arrived, and the three-label
385
+ // failure message read like a naming problem and pointed away from
386
+ // the cause. The reporter's summary — *"fallbacks are a cure for 'I
387
+ // named it wrong'; almost everything that actually failed failed
388
+ // because 'it is not there yet'"* — is the finding, and the least a
389
+ // failure can do is not mislead about which of the two it was.
390
+ const churning = await recentlyChanged(deviceQuery, options);
391
+ last.message = `none of ${tried.length} selector(s) resolved `
392
+ + `(${tried.map((t) => JSON.stringify(String(t))).join(', ')}).`
393
+ + (churning
394
+ ? ` The screen has only been still for ${churning}ms, so this is very likely timing rather than naming:`
395
+ + ' waitFor a string from the loaded state instead of adding more labels.'
396
+ : ` Last: ${last.message}`);
397
+ throw last;
398
+ }
399
+ }
247
400
  // How long this screen must hold still before it counts as settled.
248
401
  //
249
402
  // 500 ms was a constant paid by every step of every flow, and it is the
@@ -287,11 +440,24 @@ export async function runScript(
287
440
  ?? (learned && !learned.cold ? Math.max(learned.timeoutMs, floorMs) : timeoutMs);
288
441
  const settleFor = async () => {
289
442
  if (!autoSettle || !ACTION_STEPS.has(step.action)) return null;
443
+ // An action whose correct outcome is that the screen stays put cannot
444
+ // satisfy a change-based settle, so it pays the whole budget and then
445
+ // reports failure. Measured on a still screen: 1.9-2.0 seconds burned,
446
+ // `satisfied: false`, `sawChange: false`. Typing two fields on one form
447
+ // spent about four seconds waiting for transitions that were never
448
+ // going to happen, which is most of the gap a user sees between two
449
+ // fields and none of it is thinking.
450
+ //
451
+ // These steps are verified by reading the field back instead — see
452
+ // `fieldContents` — so the wait only has to cover the keystrokes
453
+ // landing, not a navigation. A short budget, and no pretence that an
454
+ // unsatisfied one means anything.
455
+ const staysPut = graph.STAYS_ON_SCREEN.has(step.action);
290
456
  const w = await api.waitFor(deviceQuery, {
291
457
  mode: 'settle',
292
458
  since: before,
293
- stableMs: stillness,
294
- timeoutMs: budgetMs,
459
+ stableMs: staysPut ? Math.min(stillness, STAYS_PUT_STILLNESS_MS) : stillness,
460
+ timeoutMs: staysPut ? STAYS_PUT_BUDGET_MS : budgetMs,
295
461
  options,
296
462
  });
297
463
  return {
@@ -303,7 +469,7 @@ export async function runScript(
303
469
  // What this wait was allowed, and where the number came from. A
304
470
  // timeout nobody can explain is how a fixed sleep comes back as a
305
471
  // constant with a comment.
306
- budgetMs,
472
+ budgetMs: graph.STAYS_ON_SCREEN.has(step.action) ? STAYS_PUT_BUDGET_MS : budgetMs,
307
473
  stillnessMs: stillness,
308
474
  quietGapMs: w.quietGapMs,
309
475
  // The baseline had already finished moving when the wait began, so it
@@ -385,22 +551,65 @@ export async function runScript(
385
551
  // happened. Without this a step that moved the screen the wrong way
386
552
  // reports success, and the flow carries on believing it worked.
387
553
  let verification = null;
554
+ // Hoisted because the note below is assembled outside the verification
555
+ // block that owns `afterScreen`. Reaching into that scope from here threw
556
+ // `afterScreen is not defined` on the very first step of a real run,
557
+ // which is what running it on a device catches and reading it does not.
558
+ let afterReading = null;
388
559
  if (verify && ACTION_STEPS.has(step.action) && beforeScreen?.hash) {
389
560
  const afterState = (await api.getState(deviceQuery, { options })).state;
390
561
  const kind = afterState.transition?.kind;
391
562
  const afterScreen = await api.screenIdentity(deviceQuery, { options, settleMs: stableMs, timeoutMs, confirmNovel });
563
+ afterReading = afterScreen;
392
564
  verification = {
393
- ...graph.verdict({ udid, prediction, before: beforeScreen.hash, after: afterScreen.hash, kind }),
565
+ ...withWayBack(
566
+ stillArriving(
567
+ belowThreshold(
568
+ graph.verdict({
569
+ udid,
570
+ prediction,
571
+ before: beforeScreen,
572
+ // The whole reading, not just its name: it carries the tokens
573
+ // that let `nearestScreen`'s similarity tolerance recognise a
574
+ // screen whose content has changed. Passing hashes here is
575
+ // what made `unexpected-screen` fire on every run that varied
576
+ // its test data.
577
+ after: afterScreen,
578
+ kind,
579
+ action: step.action,
580
+ }),
581
+ settled,
582
+ ),
583
+ afterScreen,
584
+ ),
585
+ { udid, from: beforeScreen, landed: afterScreen },
586
+ ),
394
587
  predicted: prediction ? { to: prediction.to.slice(0, 10), kind: prediction.kind, seen: prediction.count } : null,
395
588
  observed: { to: afterScreen.hash?.slice(0, 10), kind },
396
589
  };
590
+ // One more look before "nothing happened" is allowed to stand, because
591
+ // the map printed below the verdict is a later read and has three times
592
+ // contradicted it in the same response.
593
+ verification = await confirmNoChange(deviceQuery, verification, {
594
+ beforeScreen, options, stableMs, timeoutMs,
595
+ });
596
+ if (verification.lateArrival) {
597
+ // The reading the verdict was taken from is now known to be stale, so
598
+ // nothing downstream may learn a screen or an edge from it.
599
+ afterReading = verification.lateArrival;
600
+ verification.observed = { to: verification.lateArrival.hash?.slice(0, 10), kind };
601
+ delete verification.lateArrival;
602
+ }
397
603
  // Only remember what was seen on a settled screen: an edge recorded
398
604
  // mid-transition points at a screen that never really existed.
399
605
  // `confirmed` already means the fingerprint held still across two
400
606
  // independent readings, which is the thing `settled` was standing in
401
607
  // for. Requiring both meant a screen that settled slowly recorded
402
608
  // nothing at all.
403
- endScreen = afterScreen;
609
+ // The late-arrival reading when there was one, because that is the
610
+ // screen we are actually standing on — recording the stale one would
611
+ // teach the graph an edge to a screen that had already been replaced.
612
+ endScreen = afterReading ?? afterScreen;
404
613
  // An action with no observed effect teaches the graph nothing, and
405
614
  // recording it teaches something false.
406
615
  //
@@ -454,7 +663,9 @@ export async function runScript(
454
663
  const launchNote = step.action === 'launch' && settled?.noVisibleChange
455
664
  ? ' [the screen did not change, so this app was already in front — or it did not come forward]'
456
665
  : '';
666
+ const filling = stillFillingIn(afterReading?.entry);
457
667
  const note = launchNote
668
+ + (filling ? ` [settled, but ${filling} — waitFor content, do not act on this yet]` : '')
458
669
  + (settled?.smallChange ? ' [a small change, in one region only]' : '')
459
670
  + (settled?.noVisibleChange ? ' [no visible change]' : '')
460
671
  + (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
@@ -470,7 +681,33 @@ export async function runScript(
470
681
  detail: `${detail}${note}${wrongTurn ? ` [${verification.verdict}: ${verification.detail}]` : ''}`,
471
682
  settled,
472
683
  });
473
- const halt = haltDecision({ verification, stopOnUnexpected, continueOnError });
684
+ // A variant that satisfies the next step is a note, not a halt.
685
+ //
686
+ // Reported twice in one round, and it is the direct cause of two flows
687
+ // needing three calls instead of one. A tap landed on a *hash variant* of
688
+ // the screen the edge remembered — the action had plainly worked, the
689
+ // state was right, the CTA was enabled — and 26 remaining steps were
690
+ // thrown away. Variant absorption cannot help once the variant has been
691
+ // recorded as a node of its own, because absorption only claims an
692
+ // unclaimed reading, so the mismatch becomes permanent.
693
+ //
694
+ // The evidence that settles it is the flow itself: if the next step's
695
+ // target is on the screen we actually reached, we are somewhere the plan
696
+ // can continue from, whatever the fingerprint thinks. That costs one
697
+ // resolve on a path that otherwise costs a whole round trip, and it does
698
+ // not soften the verdict — the mismatch is still reported and still
699
+ // logged, because it is still the thing that found a real bug for a
700
+ // reporter twice.
701
+ const canContinue = await stillOnPlan(deviceQuery, verification, steps[i + 1], options);
702
+ const halt = haltDecision({
703
+ verification: canContinue ? { ...verification, verdict: 'unverified' } : verification,
704
+ stopOnUnexpected,
705
+ continueOnError,
706
+ });
707
+ if (canContinue) {
708
+ results[results.length - 1].detail +=
709
+ ` [landed on a variant of the expected screen; the next step resolves here, so continuing]`;
710
+ }
474
711
  if (verification?.verdict) verdicts.push(verification.verdict);
475
712
  if (metrics.ESCALATING_VERDICTS.has(verification?.verdict)) {
476
713
  noteEscalation({
@@ -478,6 +715,11 @@ export async function runScript(
478
715
  fingerprint: beforeScreen?.hash ?? null,
479
716
  reason: 'verification_failed',
480
717
  candidates: [],
718
+ // `verification_failed` is the largest reason class in the log and it
719
+ // was the only one carrying no intent, which made most of the corpus
720
+ // useless for asking what kind of decision costs us. The step knows
721
+ // what was asked for; there is no reason to drop it here.
722
+ intent: goalOf(step),
481
723
  // A halted run is a decision simframe made and stopped on; a step
482
724
  // that moved nothing carries on and leaves the judgement to whoever
483
725
  // reads the result.
@@ -505,12 +747,39 @@ export async function runScript(
505
747
  reason: why.reason,
506
748
  candidates: why.candidates,
507
749
  tried: why.tried,
750
+ // Carried from the throw site where it exists, and otherwise the step's
751
+ // own target — which is what was asked for either way.
752
+ intent: why.intent ?? goalOf(step),
508
753
  outcome: 'failed',
509
754
  wallMs: Date.now() - stepStart,
510
755
  detail: err.message,
511
756
  });
512
757
  failed = true;
513
758
  if (!continueOnError) break;
759
+ // Nothing downstream of a navigation that never happened can succeed, and
760
+ // paying its timeouts one at a time is how `--continueOnError` spent 79
761
+ // seconds on ten steps that could not possibly work — two `waitFor`s
762
+ // serving their full 9,000 ms against a screen that had not moved. What
763
+ // the operator saw was "you look stuck", and they were right.
764
+ //
765
+ // The evidence is the screen hash: consecutive failures against an
766
+ // unchanged screen are not independent attempts, they are one failure
767
+ // being re-paid. `continueOnError` means "do not stop at the first
768
+ // problem"; it does not mean "keep going after the screen has stopped
769
+ // responding to anything".
770
+ const failedOn = beforeScreen?.hash ?? null;
771
+ if (failedOn && failedOn === lastFailureScreen) {
772
+ consecutiveFailures += 1;
773
+ } else {
774
+ consecutiveFailures = 1;
775
+ lastFailureScreen = failedOn;
776
+ }
777
+ if (consecutiveFailures >= STUCK_AFTER) {
778
+ results[results.length - 1].error +=
779
+ ` — ${consecutiveFailures} consecutive failures on an unchanged screen; stopping rather than paying the remaining timeouts`;
780
+ failed = true;
781
+ break;
782
+ }
514
783
  }
515
784
  }
516
785
 
@@ -550,6 +819,10 @@ export async function runScript(
550
819
  totalMs: wallMs,
551
820
  ranSteps: results.length,
552
821
  totalSteps: steps.length,
822
+ // Every local ruling, so a wrong one is correctable rather than mysterious.
823
+ // A `stop` also carries the steps it did not attempt, so the caller resumes
824
+ // instead of re-planning.
825
+ supervisions: supervisions.length ? supervisions : undefined,
553
826
  frames,
554
827
  };
555
828
  }
@@ -564,36 +837,1214 @@ export async function runScript(
564
837
  * focus — usually fine, since a field that already had focus does not move, but
565
838
  * also exactly what a tap that missed looks like, so the caller should be told.
566
839
  */
840
+ /**
841
+ * What the field holds now, read back from the tree after the text commits.
842
+ *
843
+ * Reported three rounds running, and the third round is what forced this: the
844
+ * `type` verdict is wrong in *both* directions. A step that reported a clean
845
+ * `ok` had silently done nothing, while the step warned about as
846
+ * `no-visible-change` had landed — anti-correlated with reality, on the one
847
+ * path where acting on the warning is destructive. An operator who re-types on
848
+ * that warning doubles the field, and there is no way to clear it.
849
+ *
850
+ * A change verdict is the wrong instrument. Typing does not change the screen,
851
+ * so screen-identity movement can only ever answer a question nobody asked.
852
+ * The field's own contents answer the real one, and the tree already carries
853
+ * them — `value` has been on an AX target since MAP_VERSION 9.
854
+ *
855
+ * Accessibility only: `value` never comes from OCR, and skipping OCR keeps this
856
+ * to one cheap read. Best effort — a readback that fails must not fail the
857
+ * step, because the text may well have landed.
858
+ */
859
+ async function fieldContents(deviceQuery, target, sent, ctx) {
860
+ const wanted = alnum(sent);
861
+ // The cheap pass: the tree alone, which is authoritative whenever it answers.
862
+ const fromTree = await readbackPass(deviceQuery, target, wanted, ctx, false, FIELD_READBACK_RADIUS);
863
+ if (fromTree) return fromTree;
864
+ // The tree did not answer — and on a web view it never will. Safari's page
865
+ // content is not in the accessibility tree at all; a CI run measured
866
+ // `0 element(s) from ax` on a fully loaded page. That is not a corner case:
867
+ // **both** silent successes reported from the field were web fields, where
868
+ // this whole check has always been a no-op that printed a bare success. So
869
+ // pay for one OCR read before giving up on knowing.
870
+ //
871
+ // This pass may only ever *confirm*. OCR fuses a label with its value
872
+ // ("Telephone: 5551234567"), and on a sub-1x capture it corrupts glyphs
873
+ // ("saaiad@example.com"), so a miss here is not evidence of an empty field.
874
+ // The empty-control branch cannot fire on it either, since an OCR text
875
+ // element carries no `value` — which is the property that makes this safe.
876
+ if (!wanted) return null;
877
+ // And it has to look at a frame from *after* the keystrokes. The tree pass is
878
+ // read live and in-process, so it never had this problem; OCR runs against
879
+ // the daemon's last published frame, which on the first try was the frame
880
+ // from before the paste — so the proof arrived, the text was plainly on
881
+ // screen, and the check still said unconfirmed. Settle first, briefly.
882
+ await api.waitFor(deviceQuery, {
883
+ mode: 'settle', stableMs: 250, timeoutMs: READBACK_SETTLE_MS, options: ctx.options,
884
+ }).catch(() => null);
885
+ return readbackPass(deviceQuery, target, wanted, ctx, true, FIELD_READBACK_OCR_RADIUS);
886
+ }
887
+
888
+ /**
889
+ * One readback attempt against one sensor.
890
+ *
891
+ * `useOcr` also decides what a *miss* is allowed to mean: the tree may deny
892
+ * (an empty valued control is real evidence of absence), OCR may not.
893
+ */
894
+ async function readbackPass(deviceQuery, target, wanted, ctx, useOcr, radius) {
895
+ try {
896
+ const { entry } = await api.readScreenWith(deviceQuery, { useOcr, options: ctx.options });
897
+ const near = (entry.targets ?? []).filter(
898
+ (t) => Math.hypot((t.x ?? 0) - target.x, (t.y ?? 0) - target.y) <= radius,
899
+ );
900
+ if (!near.length) return null;
901
+ // The text may arrive as the control's `value` or as a sibling's label —
902
+ // a React Native input renders its contents as a separate text node — so
903
+ // look for what was sent across both before concluding anything. Getting
904
+ // this wrong is what made the first version of this check fail a step whose
905
+ // text was visible in the very map the failure returned.
906
+ for (const t of near) {
907
+ for (const seen of [t.value, t.label]) {
908
+ if (!seen) continue;
909
+ // `includes`, not equals: the caret is OCR'd into the value ("Maryam
910
+ // Hatami" reads back as "Maryam Hatamil"), and a long value is
911
+ // truncated by the renderer.
912
+ if (wanted && alnum(seen).includes(wanted)) {
913
+ return { value: String(seen), landed: true, focused: Boolean(t.focused) };
914
+ }
915
+ }
916
+ }
917
+ if (useOcr) return null;
918
+ // Not found. Only an *empty* valued control is evidence of absence; a
919
+ // control with no `value` attribute at all is no evidence either way.
920
+ const valued = near.find((t) => t.value != null);
921
+ if (valued && String(valued.value) === '') {
922
+ return { value: '', landed: false, focused: Boolean(valued.focused) };
923
+ }
924
+ return null;
925
+ } catch {
926
+ return null;
927
+ }
928
+ }
929
+
930
+ /**
931
+ * Journal a write, but only one that was seen to land.
932
+ *
933
+ * `back.landed` is the whole gate. An unconfirmed write is not evidence the
934
+ * value was ever in the field, and journalling one would later announce that
935
+ * text had "disappeared" when it had never arrived.
936
+ */
937
+ function journalWrite(udid, step, sent, back, ctx) {
938
+ if (!back?.landed) return;
939
+ wrote.record(udid, { selector: step.into, value: sent, screen: ctx?.screen ?? null });
940
+ }
941
+
942
+ /**
943
+ * Say when a ref was honoured by its label rather than by its number.
944
+ *
945
+ * Never silent. Re-resolving is a recovery, not a fact about the ref, and a
946
+ * recovery a caller cannot see is the shape of every silent-success bug in this
947
+ * file — so it names the number, the label it fell back to, and why.
948
+ */
949
+ export function relabelledNote(found) {
950
+ const r = found?.relabelled;
951
+ if (!r) return '';
952
+ return ` [#${r.ref} was stale, so it was re-resolved by the label it was numbered against,`
953
+ + ` ${JSON.stringify(String(r.label))} — the number was not honoured, the label was]`;
954
+ }
955
+
956
+ /** Comparison that ignores what OCR adds — a caret, a stray glyph, spacing. */
957
+ const alnum = (v) => String(v ?? '').toLowerCase().replace(/[^\p{L}\p{N}]+/gu, '');
958
+
959
+ /** How the readback is reported, and whether it contradicts what was sent. */
960
+ export function readbackNote(sent, seen) {
961
+ // No readback, or a readback that found no valued control: no evidence, and
962
+ // no evidence is not counter-evidence. The first version of this treated a
963
+ // missing `value` attribute as an empty field and failed a step whose text was
964
+ // visible in the same map the failure returned — reported, correctly, as
965
+ // worse than the verdict it replaced.
966
+ // ...but silence about it is what made two agents report a confident success
967
+ // into an empty field. Not failing is right; saying nothing is not. The note
968
+ // is the whole fix: an unconfirmed write must not read like a confirmed one.
969
+ if (!seen) {
970
+ return {
971
+ note: sent ? ' [unconfirmed — nothing on this screen reads back the field\'s contents]' : '',
972
+ empty: false,
973
+ landed: false,
974
+ };
975
+ }
976
+ if (seen.landed) {
977
+ const v = String(seen.value);
978
+ const shown = v.length > 60 ? `${v.slice(0, 60)}…` : v;
979
+ return { note: ` = ${JSON.stringify(shown)}`, empty: false, landed: true };
980
+ }
981
+ return { note: ' [the field reads empty]', empty: Boolean(sent), landed: false };
982
+ }
983
+
984
+ /**
985
+ * Is this screen still filling in?
986
+ *
987
+ * Three times in one reported run, every list in an app arrived *after* the
988
+ * settle declared the screen stable. `waitFor` on a row label fixes it, but
989
+ * that requires already knowing a string that only exists once loaded — which
990
+ * you can only learn by having failed once.
991
+ *
992
+ * Two signals, both already in hand and neither previously read. The tree often
993
+ * publishes the spinner itself: the reporter's own map contained
994
+ * `#13 element 201,263 loading`. And a list that renders its count before its
995
+ * rows announces itself — the sharpest case in that round was a `waitFor` on
996
+ * "Records" satisfied by the header **"21 Records"** while zero of the 21 rows
997
+ * existed. Even a correctly written wait can be satisfied by a promise of
998
+ * content rather than by content.
999
+ */
1000
+ const LOADING_LABEL = /^(loading|loading…|loading\.\.\.|please wait|fetching|refreshing)$/i;
1001
+ const COUNT_HEADER = /^(\d[\d,]*)\s+(records?|results?|items?|rows?|entries)\b/i;
1002
+ /**
1003
+ * The largest promised count that could be a count of rendered rows.
1004
+ *
1005
+ * Above this the header is reporting a total for a list that pages or
1006
+ * virtualises, and the number of rows on screen is unrelated to it.
1007
+ */
1008
+ const COUNT_HEADER_MAX = 30;
1009
+
1010
+ export function stillFillingIn(entry) {
1011
+ const targets = entry?.targets ?? [];
1012
+ if (!targets.length) return null;
1013
+ for (const t of targets) {
1014
+ if (LOADING_LABEL.test(String(t.label ?? '').trim())) {
1015
+ return 'a control on this screen still reads "loading"';
1016
+ }
1017
+ }
1018
+ const content = targets.filter((t) => (t.region ?? 'content') === 'content' && t.label);
1019
+ for (const t of content) {
1020
+ const m = COUNT_HEADER.exec(String(t.label).trim());
1021
+ if (!m) continue;
1022
+ const promised = Number(String(m[1]).replace(/,/g, ''));
1023
+ // Only when the promised number could plausibly be the number of *rendered*
1024
+ // rows. This is the correction to the whole idea, and it took a peer
1025
+ // ignoring the warning to see it: on a paginated or virtualised list the
1026
+ // header is a **total**, and a total says nothing at all about how many rows
1027
+ // should be on screen. Measured: `a header promises 1232 records and only 43
1028
+ // row(s) are here yet` fired on nearly every step of a list whose correct,
1029
+ // final state was 43 rendered rows. Their verdict is the one that matters —
1030
+ // *"by the fourth occurrence I was ignoring it, which is the failure mode
1031
+ // you least want from a warning."*
1032
+ //
1033
+ // A viewport holds on the order of twenty rows, so beyond a screenful the
1034
+ // count cannot be a render target and this has nothing to say.
1035
+ if (Number.isFinite(promised) && promised >= 3 && promised <= COUNT_HEADER_MAX
1036
+ && content.length < promised / 2) {
1037
+ return `a header promises ${promised} ${m[2].toLowerCase()} and only ${content.length - 1} row(s) are here yet`;
1038
+ }
1039
+ }
1040
+ return null;
1041
+ }
1042
+
1043
+ /**
1044
+ * The text a type/paste step means to send.
1045
+ *
1046
+ * `value` is the selector when `into` is present and the text when it is not,
1047
+ * and reading it as text either way put a field's own label into the field —
1048
+ * reported as `typed into "Asset*" … = "Asset*"`. Refusing beats guessing here:
1049
+ * typing a selector into a form is a wrong write, which is the one class of
1050
+ * mistake this project treats as worse than a failure.
1051
+ */
1052
+ export function textToSend(step) {
1053
+ if (step.into != null) {
1054
+ const text = step.text ?? step.value2 ?? step.with;
1055
+ if (text == null) {
1056
+ throw new Error(
1057
+ `${step.action} into ${JSON.stringify(String(step.into))} needs "text" — `
1058
+ + 'with "into" present, "value" is the selector, so there is nothing to send.',
1059
+ );
1060
+ }
1061
+ return String(text);
1062
+ }
1063
+ return String(step.text ?? step.value ?? '');
1064
+ }
1065
+
1066
+ /** The current screen hash, or null — used to notice a scroll that moved nothing. */
1067
+ async function hashNow(deviceQuery, options) {
1068
+ try {
1069
+ return (await api.screenIdentity(deviceQuery, { options, confirmNovel: false })).hash ?? null;
1070
+ } catch {
1071
+ return null;
1072
+ }
1073
+ }
1074
+
1075
+ /**
1076
+ * Failures that code can already rule on, so the model is never asked.
1077
+ *
1078
+ * Round 7 measured what happens when it is asked anyway. An element *in the
1079
+ * tree but not in view* got `wait` — while the executor's own error said, in
1080
+ * English, "waiting cannot bring it into view". An *ambiguous selector* got
1081
+ * `stop` once with the false reason "screen is elsewhere", and `wait` on a
1082
+ * later bench run. Both are deterministic: no amount of waiting or repeating
1083
+ * makes a selector unique or scrolls a viewport.
1084
+ *
1085
+ * So they are answered here, and the model's remit narrows to the one class it
1086
+ * has been reliably right about — *has this arrived yet?* On the two real
1087
+ * cases of that shape it answered correctly every time, including the one the
1088
+ * field round missed entirely.
1089
+ *
1090
+ * Narrowing a component to where it is reliable is not a workaround. It is the
1091
+ * same move as Phase 17's no-go: ask the local tier only the questions nothing
1092
+ * cheaper can answer.
1093
+ */
1094
+ export function deterministicRuling(err) {
1095
+ const m = String(err?.message ?? '');
1096
+ if (/matches \d+ things on this screen/.test(m)) {
1097
+ return { decision: 'stop', why: 'an ambiguous selector cannot be waited or retried into uniqueness — pass index, or a #ref' };
1098
+ }
1099
+ if (/in the tree but not in view/.test(m)) {
1100
+ return { decision: 'stop', why: 'waiting cannot scroll — the element needs scrollTo, or a swipe' };
1101
+ }
1102
+ if (/refused locally|destructive vocabulary/.test(m)) {
1103
+ return { decision: 'stop', why: 'a local retry may not act on this label' };
1104
+ }
1105
+ return null;
1106
+ }
1107
+
1108
+ /**
1109
+ * Ask the local supervisor whether the plan can proceed past this failure.
1110
+ *
1111
+ * Everything it needs is already in hand: the step, the failure, what is on
1112
+ * screen, how long the screen has been still, and whatever the plan told it to
1113
+ * expect. It never sees pixels and never chooses an action.
1114
+ *
1115
+ * Its own prose is deliberately not trusted as an explanation. In testing it
1116
+ * returned a correct decision with a reason citing a rule that did not apply,
1117
+ * so the decision is used and the reason is recorded — never presented to the
1118
+ * caller as the ground for what happened. Presenting a confabulated rationale
1119
+ * as fact is the mistake `seek`'s documentation already made once.
1120
+ */
1121
+ const SUPERVISOR_WAIT_MS = 4000;
1122
+ const SUPERVISOR_RETRY_MS = 900;
1123
+ /** A scroll moves at once or not at all; it does not need a transition's budget. */
1124
+ const SCROLL_SETTLE_MS = 800;
1125
+
1126
+ async function superviseFailure(deviceQuery, { goal, step, expected, err, options }) {
1127
+ if (!supervisor.requested(options)) return null;
1128
+ const settled = deterministicRuling(err);
1129
+ if (settled) return { decision: settled.decision, reason: settled.why, from: 'rule' };
1130
+ try {
1131
+ const map = await view.screenMap(deviceQuery, { options, refresh: false });
1132
+ const stillMs = map.identity?.state?.motion?.stillForMs;
1133
+ return await supervisor.judge({
1134
+ goal,
1135
+ step: `${step.action} ${JSON.stringify(String(step.value ?? step.target ?? step.into ?? step.seek ?? '').slice(0, 60))}`,
1136
+ expected,
1137
+ failure: err.message,
1138
+ screen: (map.rows ?? []).filter((r) => r.label).map((r) => r.label),
1139
+ stillMs,
1140
+ note: stillFillingIn(map.identity?.entry),
1141
+ options,
1142
+ });
1143
+ } catch {
1144
+ return null;
1145
+ }
1146
+ }
1147
+
1148
+ /**
1149
+ * How long ago the screen last moved, or null if it has been still.
1150
+ *
1151
+ * Used only to tell a timing failure from a naming failure, which is a
1152
+ * distinction every `or` chain in a reported round got wrong.
1153
+ */
1154
+ async function recentlyChanged(deviceQuery, options, withinMs = 2500) {
1155
+ try {
1156
+ const { state } = await api.getState(deviceQuery, { options });
1157
+ // `changed` is a boolean and `motion.stillForMs` is the age. Reading
1158
+ // `changed` as a timestamp is a mistake worth naming, because it would have
1159
+ // silently reported "0ms ago" on every still screen and turned this hint
1160
+ // into the opposite of information.
1161
+ if (state?.changed === true) return 0;
1162
+ if (state?.transition?.kind === 'loading') return 0;
1163
+ const still = state?.motion?.stillForMs;
1164
+ if (!Number.isFinite(still)) return null;
1165
+ return still <= withinMs ? Math.round(still) : null;
1166
+ } catch {
1167
+ return null;
1168
+ }
1169
+ }
1170
+
1171
+ /**
1172
+ * Try a step's alternatives before giving up on it.
1173
+ *
1174
+ * Four situations the owner gave, one after another, turned out to be one
1175
+ * problem: a location with no assets, a misclicked like, hunting a setting
1176
+ * through an unfamiliar menu tree, a search that returns nothing useful. *"All
1177
+ * done in maybe less than a second or a few seconds."* *"When I don't find what
1178
+ * I need somewhere I don't fall into an existential crisis. I look for it
1179
+ * somewhere else."*
1180
+ *
1181
+ * simframe answered all four the same way: the step threw, the batch was
1182
+ * abandoned, and the reasoner was asked. A step could only succeed or throw, and
1183
+ * throwing costs a round trip — `continueOnError` is all-or-nothing for a whole
1184
+ * run, which is why nobody used it.
1185
+ *
1186
+ * So a step may now carry its own fallbacks:
1187
+ *
1188
+ * {"tap": "Save", "or": ["Done", "Confirm"]}
1189
+ *
1190
+ * They are tried locally, in order, and only an exhausted list reaches the
1191
+ * model. This lengthens the *batch* instead of multiplying the round trips,
1192
+ * which is the whole objective.
1193
+ *
1194
+ * **Which failures are eligible, and why the list is short.** Only a selector
1195
+ * that did not resolve — "that label is not here, try this one". A step that
1196
+ * resolved and then went somewhere unexpected is *not* eligible: retrying from
1197
+ * the wrong screen is nonsense, and the verdict now names the way back instead.
1198
+ * Getting this wrong turns a retry primitive into a way to hammer an app until
1199
+ * something gives.
1200
+ *
1201
+ * And an alternative is simframe's own initiative, so the destructive
1202
+ * vocabulary applies to it even though it does not apply to the step the caller
1203
+ * wrote. `{"tap": "DELETE"}` is a request and is honoured; substituting
1204
+ * "DELETE" for a "Done" that did not resolve is not.
1205
+ */
1206
+ const RESOLVE_FAILURES = new Set(['unknown_screen', 'ambiguous_intent']);
1207
+
1208
+ export function alternativesFor(step) {
1209
+ const raw = step?.or ?? step?.orElse ?? step?.alternatives;
1210
+ if (raw == null) return [];
1211
+ const list = Array.isArray(raw) ? raw : [raw];
1212
+ return list.map((v) => (typeof v === 'string' ? v : v?.value ?? v?.target ?? v?.label)).filter(Boolean);
1213
+ }
1214
+
1215
+ export function mayRetryAfter(err) {
1216
+ const tagged = metrics.escalationOf(err);
1217
+ return Boolean(tagged && RESOLVE_FAILURES.has(tagged.reason));
1218
+ }
1219
+
1220
+ /**
1221
+ * Which alternatives a local retry is permitted to try, and what was refused.
1222
+ */
1223
+ export function permittedAlternatives(step, { locale } = {}) {
1224
+ const allowed = [];
1225
+ const refused = [];
1226
+ for (const label of alternativesFor(step)) {
1227
+ const verdict = vocabulary.mayActLocally(label, { locale });
1228
+ if (verdict.allowed) allowed.push(label);
1229
+ else refused.push({ label, reason: verdict.reason });
1230
+ }
1231
+ return { allowed, refused };
1232
+ }
1233
+
1234
+ /** The step to run for an alternative: the same step, aimed somewhere else. */
1235
+ export function stepWithTarget(step, label) {
1236
+ const next = { ...step };
1237
+ delete next.or;
1238
+ delete next.orElse;
1239
+ delete next.alternatives;
1240
+ for (const key of ['value', 'target', 'label', 'into']) {
1241
+ if (key in step) { next[key] = label; return next; }
1242
+ }
1243
+ next.value = label;
1244
+ return next;
1245
+ }
1246
+
1247
+ /**
1248
+ * Say the way back, in the same breath as saying we went the wrong way.
1249
+ *
1250
+ * The owner's generalisation of the recovery problem, and it is the right one:
1251
+ * *"I go to Instagram, misclick a like button — humans aren't as accurate as
1252
+ * bots. I notice immediately, I go back or I remove the like. No need to think
1253
+ * for minutes and scan the whole of Instagram's philosophy. I use what I see."*
1254
+ *
1255
+ * That recovery needs no knowledge of the app at all. It needs to notice, and
1256
+ * to know the way back — and simframe already has both. `unexpected-screen`
1257
+ * notices in about 200ms, and `graph.route` can compute a path from where we
1258
+ * landed to where we were, from edges already recorded.
1259
+ *
1260
+ * It just never said so. The step threw, the batch died, and a model round trip
1261
+ * was spent deciding something the graph could already answer. This does not
1262
+ * remove the round trip — going back changes what happens next, so a person
1263
+ * should still choose it — but it makes one round trip sufficient instead of the
1264
+ * three to six the field reports spent working out where they were.
1265
+ */
1266
+ export function wayBack(udid, from, landed, { graph: g = graph } = {}) {
1267
+ const fromHash = typeof from === 'string' ? from : from?.hash;
1268
+ const landedHash = typeof landed === 'string' ? landed : landed?.hash;
1269
+ if (!udid || !fromHash || !landedHash || fromHash === landedHash) return null;
1270
+ let path;
1271
+ try {
1272
+ path = g.route(udid, landedHash, fromHash, { maxDepth: 3 });
1273
+ } catch {
1274
+ return null;
1275
+ }
1276
+ if (!path?.length) return null;
1277
+ const steps = path.map((e) => {
1278
+ const st = e.step ?? {};
1279
+ const label = st.value ?? st.target ?? st.label ?? st.into;
1280
+ return label ? `${st.action ?? 'tap'} ${JSON.stringify(String(label).slice(0, 28))}` : (st.action ?? 'tap');
1281
+ });
1282
+ return `back to where you were: ${steps.join(' then ')}`
1283
+ + (path.length === 1 && path[0].count > 1 ? ` (seen ${path[0].count}x)` : '');
1284
+ }
1285
+
1286
+ /**
1287
+ * Do not call a screen a wrong turn while it is still arriving.
1288
+ *
1289
+ * `unexpected-screen` fired three times in one reported run and was wrong all
1290
+ * three. One cause was this: a tap applied a selection correctly and enabled
1291
+ * the submit button, but an async panel on the same screen had not come back
1292
+ * yet, so the structure differed from the settled screen the edge remembered.
1293
+ * Nothing had gone wrong; the screen was half there.
1294
+ *
1295
+ * A settle can be satisfied while content is still loading — that is Phase
1296
+ * 11.5's finding and the reason `loading` exists — so the two must be read
1297
+ * together. An incomplete screen cannot contradict a prediction, and
1298
+ * `unverified` is the honest verdict: we do not know yet.
1299
+ *
1300
+ * The other cause was data variation, which this does not address: picking a
1301
+ * different test asset changes the content and the check reads it as a wrong
1302
+ * turn. That needs structural comparison and is filed, not fixed.
1303
+ */
1304
+ export function withWayBack(verification, { udid, from, landed, graph: g } = {}) {
1305
+ if (verification?.verdict !== 'unexpected-screen') return verification;
1306
+ const route = wayBack(udid, from, landed, g ? { graph: g } : undefined);
1307
+ return route ? { ...verification, detail: `${verification.detail} — ${route}` } : verification;
1308
+ }
1309
+
1310
+ export function stillArriving(verification, afterScreen) {
1311
+ if (verification?.verdict !== 'unexpected-screen') return verification;
1312
+ if (afterScreen?.loading !== true) return verification;
1313
+ return {
1314
+ ...verification,
1315
+ verdict: 'unverified',
1316
+ detail: 'the screen this reached is still loading, so it cannot be compared yet'
1317
+ + ' — re-read it, or waitFor the content you expect, before treating this as a wrong turn',
1318
+ };
1319
+ }
1320
+
1321
+ /**
1322
+ * Can the plan continue from where we actually landed?
1323
+ *
1324
+ * Only asked when the verdict is `unexpected-screen`, and only answered by the
1325
+ * next step's own selector resolving here. A screen that can serve the next step
1326
+ * is not a wrong turn in any sense the caller cares about.
1327
+ */
1328
+ async function stillOnPlan(deviceQuery, verification, nextStep, options) {
1329
+ if (verification?.verdict !== 'unexpected-screen' || !nextStep) return false;
1330
+ const target = nextStep.value ?? nextStep.target ?? nextStep.label ?? nextStep.into;
1331
+ // A coordinate resolves anywhere and a ref was numbered on another screen, so
1332
+ // neither is evidence about where we are.
1333
+ if (!target || /^#\d+$/.test(String(target).trim()) || /^@?-?\d+\s*,\s*-?\d+$/.test(String(target).trim())) return false;
1334
+ if (!ACTION_STEPS.has(nextStep.action) && nextStep.action !== 'assert' && nextStep.action !== 'waitFor') return false;
1335
+ try {
1336
+ const hit = await api.locate(deviceQuery, String(target), { options });
1337
+ return Boolean(hit?.target);
1338
+ } catch {
1339
+ return false;
1340
+ }
1341
+ }
1342
+
1343
+ /**
1344
+ * Reconcile the two sensors when they disagree about "did anything happen".
1345
+ *
1346
+ * The wait watches regions and the verdict compares screen identity, so a tap
1347
+ * that moved one cell produced `[a small change, in one region only]` and
1348
+ * `no-visible-change: the screen did not change` four lines apart. Both were
1349
+ * true of different questions, and the pair reads as a contradiction rather
1350
+ * than as a measurement.
1351
+ *
1352
+ * The verdict stands — a change too small to move the screen's identity is the
1353
+ * finding — but it should say what was actually observed.
1354
+ */
1355
+ /**
1356
+ * How long a "nothing happened" verdict waits before believing itself.
1357
+ *
1358
+ * Short, because it is paid on a path that is currently often wrong: three
1359
+ * `no-visible-change` verdicts in one field round were contradicted by the
1360
+ * element list printed directly beneath them.
1361
+ */
1362
+ const LATE_CHANGE_MS = 700;
1363
+
1364
+ /**
1365
+ * Give a "nothing happened" verdict one more look before it stands.
1366
+ *
1367
+ * The verdict reads the screen's identity, and the map underneath it is a
1368
+ * separate, later read — so a web view or a slow list can render in between,
1369
+ * and then the response contradicts itself. Reported from the field, three
1370
+ * times in one round: `no-visible-change: the screen did not change` above a
1371
+ * map whose hash had moved `46a0a26d → 1a8e8be1` and which listed three
1372
+ * dropdown options that had just appeared. The agent's summary of the cost is
1373
+ * the reason this is worth a re-read: *"I was instructed to spend a call
1374
+ * disproving a claim the same response had already disproved."*
1375
+ *
1376
+ * It is also the verdict that must not be wrong in this direction. It escalates
1377
+ * — the hint line tells the caller to stop and check — so a false one buys a
1378
+ * round trip every time, which is exactly what this project is built to remove.
1379
+ *
1380
+ * The test is provable rather than heuristic: if the screen is now different
1381
+ * from how it was *before* the action, the action changed it.
1382
+ */
1383
+ async function confirmNoChange(deviceQuery, verification, { beforeScreen, options, stableMs, timeoutMs }) {
1384
+ if (verification?.verdict !== 'no-visible-change' || !beforeScreen?.hash) return verification;
1385
+ await api.waitFor(deviceQuery, {
1386
+ mode: 'settle', stableMs: 250, timeoutMs: LATE_CHANGE_MS, options,
1387
+ }).catch(() => null);
1388
+ const again = await api.screenIdentity(deviceQuery, { options, settleMs: stableMs, timeoutMs }).catch(() => null);
1389
+ if (!again?.hash || again.hash === beforeScreen.hash) return verification;
1390
+ return {
1391
+ ...verification,
1392
+ verdict: 'unverified',
1393
+ detail: 'the screen did change, but not until after the verdict had been taken'
1394
+ + ` (${beforeScreen.hash.slice(0, 8)} → ${again.hash.slice(0, 8)})`
1395
+ + ' — a web view or a slow list can render after a settle has reported it still',
1396
+ lateArrival: again,
1397
+ };
1398
+ }
1399
+
1400
+ export function belowThreshold(verification, settled) {
1401
+ if (verification?.verdict !== 'no-visible-change' || !settled?.smallChange) return verification;
1402
+ return {
1403
+ ...verification,
1404
+ detail: 'the screen changed in one region only, by too little to be a different screen'
1405
+ + ' — if that was the whole effect, this is fine; if a transition was expected, it did not happen',
1406
+ };
1407
+ }
1408
+
1409
+ /**
1410
+ * Is the field focused? One read, and the answer is advisory.
1411
+ *
1412
+ * Filling one field used to be verified five times: the field exists, it took
1413
+ * focus, the text landed, the screen settled, the screen is still the screen.
1414
+ * None of that involves a model — it is all local — which is why several
1415
+ * seconds could pass between two fields with no thinking in them at all.
1416
+ *
1417
+ * Two of the five were change-based waits, and an action that changes nothing
1418
+ * cannot satisfy one: measured on a still screen, 1.9-2.0 seconds each, both
1419
+ * returning `satisfied: false`. Tapping into a text field barely moves the
1420
+ * screen, and with a hardware keyboard attached to the simulator no software
1421
+ * keyboard appears, so there is often nothing to see.
1422
+ *
1423
+ * The collapse is to verify the *result* rather than each precondition. If the
1424
+ * text landed, focus obviously worked, so checking focus first is redundant
1425
+ * with checking the outcome. This stays as a single ~85 ms accessibility read
1426
+ * because typing into an unfocused field loses the keystrokes, so it is worth
1427
+ * one cheap look — but it never blocks and it never fails the step. The
1428
+ * readback after typing decides.
1429
+ */
1430
+ async function focusHint(deviceQuery, target, ctx) {
1431
+ try {
1432
+ const { entry } = await api.readScreenWith(deviceQuery, { useOcr: false, options: ctx.options });
1433
+ let elsewhere = false;
1434
+ for (const t of entry.targets ?? []) {
1435
+ if (t.focused !== true) continue;
1436
+ if (Math.hypot((t.x ?? 0) - target.x, (t.y ?? 0) - target.y) <= FIELD_READBACK_RADIUS) {
1437
+ return { focused: true, elsewhere: false };
1438
+ }
1439
+ elsewhere = true;
1440
+ }
1441
+ return { focused: false, elsewhere };
1442
+ } catch {
1443
+ return { focused: false, elsewhere: false };
1444
+ }
1445
+ }
1446
+
1447
+ /**
1448
+ * How long to wait for the tap to become focus before inserting text.
1449
+ *
1450
+ * Paid only when the tree has not yet said the field is focused, so a field
1451
+ * that focuses instantly costs one read exactly as before.
1452
+ */
1453
+ const FOCUS_WAIT_MS = 700;
1454
+ const FOCUS_POLL_MS = 80;
1455
+
1456
+ /**
1457
+ * Wait for focus to arrive, without treating silence as failure.
1458
+ *
1459
+ * External research settled a symptom two agents reported independently and
1460
+ * neither could reproduce: one insertion primitive returns `ok` into an empty
1461
+ * field, and the other then works. It is a **focus race**, not a defect in
1462
+ * either primitive — a keystroke delivered before a web view commits focus to
1463
+ * its input is simply dropped, and WDA copes by checking `hasKeyboardFocus`
1464
+ * before typing. So the investigation everyone reached for, paste versus type,
1465
+ * was the wrong one and would never have converged.
1466
+ *
1467
+ * The wait ends the moment the tree names our target as focused. If the budget
1468
+ * runs out with the tree saying nothing, we insert anyway — because "the tree
1469
+ * is silent about focus" has never been evidence the tap missed, and this file
1470
+ * has been wrong in that direction before.
1471
+ */
1472
+ async function awaitFocus(deviceQuery, target, ctx) {
1473
+ const deadline = Date.now() + FOCUS_WAIT_MS;
1474
+ let last = await focusHint(deviceQuery, target, ctx);
1475
+ while (!last.focused && Date.now() < deadline) {
1476
+ await new Promise((r) => setTimeout(r, FOCUS_POLL_MS));
1477
+ last = await focusHint(deviceQuery, target, ctx);
1478
+ }
1479
+ return last;
1480
+ }
1481
+
567
1482
  async function focusField(deviceQuery, udid, step, ctx) {
568
1483
  const found = await api.locate(deviceQuery, step.into, { index: step.index, refresh: step.refresh });
569
- // What this field has cost to focus before, on this screen. Cold, or with no
570
- // verification running, that is exactly the three constants above; measured,
571
- // it can only be longer. `graph.focusPlan` carries the reason it is either.
572
- const plan = ctx.focus?.plan ?? {
573
- reactionMs: FOCUS_REACTION_MS, timeoutMs: FOCUS_TIMEOUT_MS, cold: true, from: 'no timing in hand',
574
- };
575
1484
  const tappedAt = Date.now();
576
1485
  await input.tapPoint(udid, found.target.x, found.target.y);
577
- const focused = await api.waitFor(deviceQuery, {
578
- mode: 'settle',
579
- stableMs: FOCUS_STABLE_MS,
580
- reactionMs: plan.reactionMs,
581
- timeoutMs: plan.timeoutMs,
582
- options: ctx.options,
583
- });
584
- // Only a wait that was satisfied is a measurement of how long focus takes. A
585
- // reaction window that ran out measures how long we were prepared to watch a
586
- // screen that did not move, and banking that would teach the edge the cost of
587
- // its own impatience — the estimator mistake learned stillness made.
588
- if (ctx.focus && focused.satisfied) ctx.focus.observedMs = Date.now() - tappedAt;
1486
+ const focused = await awaitFocus(deviceQuery, found.target, ctx);
1487
+ // There IS a focus wait again, and this comment used to say there was not.
1488
+ // It was removed when one accessibility read replaced it, which was right for
1489
+ // native fields and wrong for web views: a keystroke delivered before a web
1490
+ // view commits focus is dropped, and that is the whole explanation of the
1491
+ // intermittent silent write two agents reported. The wait is bounded, it ends
1492
+ // the moment the tree names the target as focused, and silence still means
1493
+ // proceed — see `awaitFocus`.
1494
+ //
1495
+ // The focus *distribution* is still not collected, and that part stands:
1496
+ // banking the duration of these polls under the old name would keep a number
1497
+ // nobody uses, measuring something other than what its name says, which is
1498
+ // the shape of the learned-stillness mistake. `graph.focusPlan` and the
1499
+ // `focusSamples` it reads stay in place, unfed; if nothing claims them they
1500
+ // should go.
1501
+ void tappedAt;
589
1502
  return {
590
1503
  found,
591
1504
  where: `"${found.target.label}" at ${found.target.x},${found.target.y}`,
592
- quiet: focused.satisfied ? '' : ' [the field did not visibly take focus]',
593
- waited: focused.satisfied && !plan.cold ? ` [focus in ${focused.waitedMs}ms, ${plan.from}]` : '',
1505
+ // Only claimed when the tree named a *different* focused element. "The
1506
+ // tree says nothing about focus" is not evidence the tap missed, and
1507
+ // asserting it from a screen that simply did not move is what made this
1508
+ // note wrong on a correctly focused field.
1509
+ // Only claimed when the tree named a *different* focused element. Silence
1510
+ // about focus is not evidence the tap missed, and asserting it from a
1511
+ // screen that simply did not move is what made this note wrong on a
1512
+ // correctly focused field.
1513
+ quiet: focused.elsewhere && !focused.focused ? ' [focus is on another element of this screen]' : '',
594
1514
  };
595
1515
  }
596
1516
 
1517
+ /**
1518
+ * Look for something that is not on this screen, the way a person does.
1519
+ *
1520
+ * The owner's description, and it is the behaviour this implements: *"I am in a
1521
+ * new app's settings, I look for something like change username. I go to each
1522
+ * menu, check the items, nothing like that? Next menu, until I find it."* And:
1523
+ * *"when I don't find what I need somewhere, I don't fall into an existential
1524
+ * crisis. I look for it somewhere else."*
1525
+ *
1526
+ * Today a miss is an existential crisis — nothing resolves, the step throws, the
1527
+ * batch dies, and the reasoner is asked. Twelve of 173 escalations, each one
1528
+ * stopping a batch, so a five-menu hunt costs ten or more round trips for
1529
+ * something a person does in seconds.
1530
+ *
1531
+ * `seek` opens containers, checks, and comes back, inside a hard budget.
1532
+ *
1533
+ * **It acts, and saying otherwise is what made it dangerous.** The first
1534
+ * version of this comment said it "finds and does not act", meaning it does not
1535
+ * tap the *target* — but opening a door is an action, doors change state, and a
1536
+ * reader who trusted that sentence handed `seek` a flow it could destroy. It
1537
+ * did: it opened CANCEL, then AI TROUBLESHOOTING, then pressed "YES, THIS FIXED
1538
+ * MY PROBLEM", ending five screens deep in a live support chat with a
1539
+ * half-completed service request gone. One label further along was SUBMIT
1540
+ * SERVICE REQUEST.
1541
+ *
1542
+ * What is true: it does not tap the target — it leaves you on the screen where
1543
+ * the target resolves and says so, and the caller taps it as the next step of
1544
+ * the same batch. What it *does* tap is doors, filtered by the exploration
1545
+ * vocabulary (`vocabulary.openableAsDoor`), which is much stricter than the
1546
+ * substitution list and refuses anything that commits, abandons, answers or
1547
+ * leaves. And it returns to the screen it started from before handing back,
1548
+ * saying plainly when it could not.
1549
+ *
1550
+ * Ordering is the pluggable part, and the only part a local model touches. With
1551
+ * `SIMFRAME_PLANNER` unset the order is mechanical — the screen's own reading
1552
+ * order — and every candidate gets tried anyway; the model only changes which
1553
+ * comes first. That is why it is safe to try and why it is A/B-testable: run the
1554
+ * same flow with the flag off and on and compare steps to target.
1555
+ */
1556
+ export const SEEK_BUDGET = 6;
1557
+
1558
+ /** How many candidates a ranker is asked about, and how long a door gets to open. */
1559
+ export const SEEK_RANK_CANDIDATES = 12;
1560
+ /** How many back-steps a failed seek may spend returning to where it began. */
1561
+ const RETURN_BUDGET = 8;
1562
+ /** How long a disagreeing assert waits before looking again. */
1563
+ const ASSERT_RETAKE_MS = 900;
1564
+ /** Consecutive failures on one unchanged screen before a continue-on-error run gives up. */
1565
+ export const STUCK_AFTER = 3;
1566
+ const SEEK_SETTLE_MS = 1200;
1567
+
1568
+ async function candidatesToOpen(deviceQuery, udid, { visited, options }) {
1569
+ const map = await view.screenMap(deviceQuery, { options, refresh: true });
1570
+ const here = map.identity?.hash ?? null;
1571
+ const labels = [];
1572
+ for (const r of map.rows ?? []) {
1573
+ // Content only. A nav-bar title is not a door — the first version of this
1574
+ // opened "Settings", which is the name of the screen it was already on.
1575
+ if ((r.region ?? 'content') !== 'content') continue;
1576
+ if (!r.label || r.enabled === false) continue;
1577
+ if (visited.has(String(r.label))) continue;
1578
+ // Permissive about *shape*, strict about *vocabulary* — and that pairing is
1579
+ // the correction, not a loosening. Requiring `actsInteractive` found **zero
1580
+ // doors** on a screen holding two real pickers, because a React Native
1581
+ // picker is a generic element with no value and the tree has been wrong
1582
+ // about roles in every round. Meanwhile the door vocabulary was too loose
1583
+ // and opened CANCEL. The tree is unreliable about what is tappable and the
1584
+ // label is reliable about what must not be opened, so trust each where it
1585
+ // is trustworthy.
1586
+ if (!vocabulary.openableAsDoor(r.label)) continue;
1587
+ // A paragraph is not a door.
1588
+ if (String(r.label).length > 48) continue;
1589
+ labels.push(String(r.label));
1590
+ }
1591
+ return { here, map, labels: [...new Set(labels)] };
1592
+ }
1593
+
1594
+ /**
1595
+ * Get back to the screen above, by whatever means this app offers.
1596
+ *
1597
+ * This is where `seek` first stranded itself, on a finding already in
1598
+ * DEFERRED: the nav-bar back chevron is invisible to simframe, so
1599
+ * `locate("back")` throws and one door was all it ever opened. The left-edge
1600
+ * gesture needs no label and works on any pushed screen, so it is the fallback
1601
+ * rather than the exception — and the return is confirmed, because a swipe that
1602
+ * did nothing would make the next candidate a tap on a screen we did not mean
1603
+ * to be on.
1604
+ */
1605
+ async function goBack(deviceQuery, udid, step, ctx, { from, to }) {
1606
+ const byLabel = await (async () => {
1607
+ try {
1608
+ const route = from && to ? graph.route(udid, from, to, { maxDepth: 1 }) : null;
1609
+ const st = route?.length === 1 ? route[0].step ?? {} : {};
1610
+ const label = st.value ?? st.target ?? st.label;
1611
+ return label && vocabulary.actableLocally(label) ? String(label) : 'back';
1612
+ } catch {
1613
+ return 'back';
1614
+ }
1615
+ })();
1616
+ let acted = false;
1617
+ try {
1618
+ const b = await api.locate(deviceQuery, byLabel, { options: ctx.options });
1619
+ await input.tapPoint(udid, b.target.x, b.target.y);
1620
+ acted = true;
1621
+ } catch { /* no visible back control — use the gesture */ }
1622
+ if (!acted) {
1623
+ try {
1624
+ const geo = await ctx.screen();
1625
+ const y = Math.round((geo.pointHeight ?? 874) / 2);
1626
+ await input.swipe(udid, { x: 2, y }, { x: Math.round((geo.pointWidth ?? 402) * 0.6), y }, { durationMs: 250 });
1627
+ acted = true;
1628
+ } catch {
1629
+ return false;
1630
+ }
1631
+ }
1632
+ await api.waitFor(deviceQuery, { mode: 'settle', stableMs: step.stableMs ?? 400, timeoutMs: step.timeoutMs ?? SEEK_SETTLE_MS, options: ctx.options });
1633
+ try {
1634
+ const now = await api.screenIdentity(deviceQuery, { options: ctx.options, confirmNovel: false });
1635
+ return Boolean(now.hash) && now.hash !== from;
1636
+ } catch {
1637
+ return false;
1638
+ }
1639
+ }
1640
+
1641
+ async function seek(deviceQuery, udid, step, ctx) {
1642
+ const goal = step.seek ?? step.value ?? step.target;
1643
+ if (!goal) throw new Error('usage: {"seek": "what you are looking for"}');
1644
+ const budget = Math.max(1, Math.min(step.budget ?? SEEK_BUDGET, 12));
1645
+ const options = ctx.options;
1646
+
1647
+ const found = async () => {
1648
+ try {
1649
+ const r = await api.locate(deviceQuery, String(goal), { options });
1650
+ return r?.target ? r : null;
1651
+ } catch {
1652
+ return null;
1653
+ }
1654
+ };
1655
+
1656
+ const already = await found();
1657
+ if (already) return `"${goal}" is already here: "${already.target.label}" at ${already.target.x},${already.target.y}`;
1658
+
1659
+ // Keyed on the label alone, not label-per-screen. Keying it per screen let it
1660
+ // cycle — General, About, General, Screen Capture, General, About — because a
1661
+ // list's identity is not perfectly stable across a return to it, so the same
1662
+ // door read as unvisited. Within one seek, one attempt per label is enough.
1663
+ const visited = new Set();
1664
+ const opened = [];
1665
+ const trail = [];
1666
+ let spent = 0;
1667
+ const origin = await (async () => {
1668
+ try {
1669
+ return (await api.screenIdentity(deviceQuery, { options, confirmNovel: false })).hash ?? null;
1670
+ } catch {
1671
+ return null;
1672
+ }
1673
+ })();
1674
+
1675
+ // Depth-first, because that is what a person does: Accessibility, then
1676
+ // Display & Text Size, then Larger Text. The first version always came back
1677
+ // after one probe and could never reach anything two levels down, which is
1678
+ // where most settings live.
1679
+ while (spent < budget) {
1680
+ const { here, labels } = await candidatesToOpen(deviceQuery, udid, { visited, options });
1681
+ if (!labels.length) {
1682
+ // Nothing new here. Back out to the screen above and try its next door.
1683
+ if (!trail.length) break;
1684
+ const to = trail.pop();
1685
+ if (!await goBack(deviceQuery, udid, step, ctx, { from: here, to })) break;
1686
+ continue;
1687
+ }
1688
+ // Only the first handful go to the ranker. A forty-label prompt costs more
1689
+ // to answer and the budget will never reach the tail anyway; and the model
1690
+ // is asked about candidates, not given the screen.
1691
+ const asked = labels.slice(0, SEEK_RANK_CANDIDATES);
1692
+ const ranked = await planner.rank(String(goal), asked, { deviceOptions: options });
1693
+ const ordered = ranked ? [...ranked, ...labels.slice(SEEK_RANK_CANDIDATES)] : labels;
1694
+ const pick = ordered[0];
1695
+ visited.add(pick);
1696
+ spent += 1;
1697
+
1698
+ try {
1699
+ const door = await api.locate(deviceQuery, pick, { options });
1700
+ await input.tapPoint(udid, door.target.x, door.target.y);
1701
+ } catch {
1702
+ continue; // a label that will not resolve is not a door
1703
+ }
1704
+ await api.waitFor(deviceQuery, { mode: 'settle', stableMs: step.stableMs ?? 400, timeoutMs: step.timeoutMs ?? SEEK_SETTLE_MS, options });
1705
+
1706
+ let landed = null;
1707
+ try {
1708
+ landed = (await api.screenIdentity(deviceQuery, { options, confirmNovel: false })).hash;
1709
+ } catch { /* unknown where we are; the found() check still decides */ }
1710
+ // A door that led nowhere is not a door, and descending would corrupt the
1711
+ // trail with a screen we never left.
1712
+ if (landed && landed !== here) trail.push(here);
1713
+
1714
+ const hit = await found();
1715
+ if (hit) {
1716
+ return `found "${goal}" as "${hit.target.label}" at ${hit.target.x},${hit.target.y}`
1717
+ + ` after opening ${opened.concat(pick).map((l) => JSON.stringify(l)).join(' -> ')}`
1718
+ + ` (${spent} of ${budget} step(s)${planner.requested(options) ? ', planner-ordered' : ''})`;
1719
+ }
1720
+ opened.push(pick);
1721
+ }
1722
+
1723
+ // Come back before handing over.
1724
+ //
1725
+ // The contract is depth-first *with return*, and on failure it was not: a
1726
+ // reported run ended five doors deep in a live support chat, on an unrelated
1727
+ // screen, with the flow it started from destroyed. Leaving a caller somewhere
1728
+ // they did not ask to be is worse than failing, because everything they try
1729
+ // next is aimed at the wrong screen.
1730
+ let restored = origin == null;
1731
+ for (let i = 0; i < RETURN_BUDGET && !restored; i += 1) {
1732
+ let now = null;
1733
+ try {
1734
+ now = (await api.screenIdentity(deviceQuery, { options, confirmNovel: false })).hash;
1735
+ } catch { break; }
1736
+ if (now === origin) { restored = true; break; }
1737
+ if (!await goBack(deviceQuery, udid, step, ctx, { from: now, to: origin })) break;
1738
+ }
1739
+
1740
+ // Say where we got to and what is there, not just that we failed.
1741
+ //
1742
+ // Measured on the first real run of this: with the ranker on, `seek "make the
1743
+ // text bigger"` went Accessibility -> Display & Text Size in two steps — the
1744
+ // right place — and then reported "not found", because the control is called
1745
+ // "Larger Text" and `locate` matches labels lexically. The navigation was
1746
+ // right and the arrival was unreportable, so the caller learned nothing from a
1747
+ // 20-second search.
1748
+ //
1749
+ // The whole point of a local tier is to make ONE round trip sufficient. So the
1750
+ // hand-back carries the landing: where we are, and what is on it.
1751
+ const landing = await (async () => {
1752
+ try {
1753
+ const map = await view.screenMap(deviceQuery, { options, refresh: false });
1754
+ const here = (map.rows ?? [])
1755
+ .filter((r) => (r.region ?? 'content') === 'content' && r.label)
1756
+ .slice(0, 12)
1757
+ .map((r) => JSON.stringify(String(r.label).slice(0, 32)));
1758
+ return here.length ? ` Now on ${map.name ? `"${map.name}"` : (map.identity?.hash ?? 'an unnamed screen').slice(0, 8)}, which offers: ${here.join(', ')}.` : '';
1759
+ } catch {
1760
+ return '';
1761
+ }
1762
+ })();
1763
+ throw metrics.tag(
1764
+ new Error(
1765
+ `"${goal}" did not resolve by label within ${spent} of ${budget} steps.`
1766
+ + (opened.length ? ` Opened: ${opened.map((l) => JSON.stringify(l)).join(' -> ')}.` : ' Nothing here looked like a container.')
1767
+ + landing
1768
+ + (restored
1769
+ ? ' Back on the screen you started from.'
1770
+ : ' **You are not back where you started** — the way back could not be found, so read the screen before acting.')
1771
+ + ' If one of those is what you meant, tap it by name; otherwise say where to look or raise the budget.',
1772
+ ),
1773
+ 'no_plan',
1774
+ { intent: String(goal), tried: opened },
1775
+ );
1776
+ }
1777
+
1778
+ /**
1779
+ * How far the screen actually moved, measured from the elements themselves.
1780
+ *
1781
+ * The missing sensor, and the owner named the need exactly: *"you have to find a
1782
+ * way to detect where you currently are, so when you are already on top you
1783
+ * don't scroll further, or the bottom."* Nothing on either platform reports a
1784
+ * scroll offset, so it has to be inferred — and the two signals tried before
1785
+ * this both failed on a real page.
1786
+ *
1787
+ * The **screen hash** changes forever on a page whose footer has live content,
1788
+ * so "the hash stopped changing" never fired and a sweep thrashed at the bottom
1789
+ * for forty seconds. **New labels** fail the other way: a gesture that reveals
1790
+ * only a little adds nothing new and looks like an end, which stopped a sweep
1791
+ * two sections above the form it was looking for.
1792
+ *
1793
+ * Element geometry answers it directly. Take the labels present both before and
1794
+ * after, and compare their y. Unchanged means nothing moved — that is an end,
1795
+ * whatever the hash or the label set says. A negative median delta means the
1796
+ * content came up, so we travelled down, and by how much.
1797
+ */
1798
+ export const SCROLL_STILL_PX = 6;
1799
+
1800
+ export function scrollDelta(before, after) {
1801
+ // Content only. Fixed chrome is the trap: a browser's bottom toolbar is five
1802
+ // elements whose y never changes, and with few shared content rows they drag
1803
+ // the median to zero — so a page that had plainly scrolled measured as
1804
+ // motionless and a sweep declared the bottom after one section. Only things
1805
+ // that can move are evidence that something moved.
1806
+ const scrolls = (r) => {
1807
+ const region = r?.region ?? 'content';
1808
+ return region !== 'nav-bar' && region !== 'tab-bar' && region !== 'status-bar' && region !== 'keyboard';
1809
+ };
1810
+ const was = new Map();
1811
+ for (const r of before ?? []) {
1812
+ if (!scrolls(r)) continue;
1813
+ const k = alnum(r.label);
1814
+ if (k && Number.isFinite(r.y) && !was.has(k)) was.set(k, r.y);
1815
+ }
1816
+ const deltas = [];
1817
+ for (const r of after ?? []) {
1818
+ if (!scrolls(r)) continue;
1819
+ const k = alnum(r.label);
1820
+ if (!k || !Number.isFinite(r.y) || !was.has(k)) continue;
1821
+ deltas.push(r.y - was.get(k));
1822
+ }
1823
+ if (!deltas.length) {
1824
+ // Nothing in common. Either everything changed — a real move — or the read
1825
+ // failed; either way this is not evidence of an end.
1826
+ return { moved: null, px: null, shared: 0 };
1827
+ }
1828
+ deltas.sort((a, b) => a - b);
1829
+ const median = deltas[deltas.length >> 1];
1830
+ return {
1831
+ moved: Math.abs(median) > SCROLL_STILL_PX,
1832
+ px: Math.round(median),
1833
+ shared: deltas.length,
1834
+ };
1835
+ }
1836
+
1837
+ /**
1838
+ * Sweep a scrollable screen section by section, and fill what is there.
1839
+ *
1840
+ * The owner's algorithm: *"to scan a scrollable screen and fill, you need to
1841
+ * detect min/max scroll and look at it section by section. Section 1: anything
1842
+ * to fill? Do it. Not? Scroll to section 2."*
1843
+ *
1844
+ * That is the right shape for a reason `scrollTo` cannot fix. The tree publishes
1845
+ * what is rendered, so a long form is only ever knowable in pieces — and one
1846
+ * gesture travels a non-deterministic distance (four rows once, one row the
1847
+ * next), so "jump to section 3" is not a thing that exists. Acting on whatever
1848
+ * the current viewport holds is the only plan that survives that.
1849
+ *
1850
+ * **Both ends are detected, and by the right signal.** The first version keyed
1851
+ * on the screen hash and ran its whole budget on a page whose footer has live
1852
+ * content: the hash kept changing, so "stopped moving" never fired, and the
1853
+ * operator watched it thrash at the bottom for forty seconds. A section that
1854
+ * contributes **no new elements** is the end, whatever the hash says.
1855
+ *
1856
+ * **And it starts at the beginning**, because a sweep from an unknown position
1857
+ * covers an unknown amount — the same forty seconds began with section 1 being
1858
+ * the page footer. Going up is bounded and still wants two stalls, but the
1859
+ * gesture that confirms the second one is a 60pt nudge rather than a full
1860
+ * section: a section-sized up-swipe at the top of a web page is
1861
+ * pull-to-refresh, and a reload clears the form the sweep exists to fill.
1862
+ */
1863
+ export const SWEEP_SECTIONS = 10;
1864
+
1865
+ const sweepKey = (r) => `${alnum(r.label)}\u0000${Math.round((r.x ?? 0) / 8)}`;
1866
+
1867
+ async function sectionHere(deviceQuery, options) {
1868
+ try {
1869
+ const map = await view.screenMap(deviceQuery, { options, refresh: true });
1870
+ return (map.rows ?? []).filter((r) => r.label);
1871
+ } catch {
1872
+ return [];
1873
+ }
1874
+ }
1875
+
1876
+ /**
1877
+ * Advance by a section, not by whatever a default swipe happens to do.
1878
+ *
1879
+ * Measured on a real page: `{"scroll":"down"}` moved **28, 42 and 58 points** on
1880
+ * an 874-point screen — about five per cent of a viewport per gesture. Covering
1881
+ * a page that way takes dozens of swipes, which is precisely what the operator
1882
+ * was watching: *"I still see scroll thrashing, you scrolled too much."* Too
1883
+ * many gestures, not too far each.
1884
+ *
1885
+ * A section is a viewport. `SECTION_FRACTION` leaves a band of overlap so
1886
+ * nothing falls between two reads, which is the whole reason to sweep rather
1887
+ * than to jump.
1888
+ */
1889
+ const SECTION_FRACTION = 0.7;
1890
+
1891
+ /**
1892
+ * How far the gesture that *confirms* the top travels.
1893
+ *
1894
+ * Small on purpose. iOS Safari's pull-to-refresh fires on overscroll distance,
1895
+ * and a section-sized up-swipe drags roughly 600 points — at the top that is an
1896
+ * enormous overscroll and it reloads the page. Sixty points still moves a page
1897
+ * that has anywhere left to go, which is all the confirmation needs to do, and
1898
+ * stays well under the refresh threshold when it does not.
1899
+ */
1900
+ const TOP_CONFIRM_PT = 60;
1901
+
1902
+ async function scrollOne(deviceQuery, udid, dir, ctx, { spanPt } = {}) {
1903
+ const geo = await ctx.screen();
1904
+ const h = geo?.pointHeight ?? 874;
1905
+ const x = Math.round((geo?.pointWidth ?? 402) / 2);
1906
+ const span = spanPt ? Math.round(spanPt) : Math.round(h * SECTION_FRACTION);
1907
+ const top = Math.round(h * 0.12);
1908
+ const from = dir === 'up' ? { x, y: top } : { x, y: top + span };
1909
+ const to = dir === 'up' ? { x, y: top + span } : { x, y: top };
1910
+ try {
1911
+ await input.swipe(udid, from, to, { durationMs: 260 });
1912
+ } catch {
1913
+ // A platform without a swipe still has a scroll.
1914
+ await runStep(deviceQuery, udid, { action: 'scroll', value: dir }, ctx);
1915
+ }
1916
+ await api.waitFor(deviceQuery, {
1917
+ mode: 'stable', stableMs: 200, timeoutMs: SCROLL_SETTLE_MS, options: ctx.options,
1918
+ }).catch(() => null);
1919
+ }
1920
+
1921
+ async function sweep(deviceQuery, udid, step, ctx) {
1922
+ const limit = Math.max(1, Math.min(step.sections ?? SWEEP_SECTIONS, 20));
1923
+ const options = ctx.options;
1924
+ const fill = step.fill && typeof step.fill === 'object' ? { ...step.fill } : null;
1925
+ const wanted = typeof step.sweep === 'string' && step.sweep !== 'all' ? step.sweep.trim() : null;
1926
+
1927
+ // To the beginning, unless told otherwise. Bounded, and it still wants two
1928
+ // stalls before believing the top — a sticky element inside a page can make a
1929
+ // single median read as zero, which is why one stall used to send it from the
1930
+ // top straight to the end without reading the form between.
1931
+ //
1932
+ // What changed is the gesture that buys the second stall. It used to be
1933
+ // another section-sized up-swipe, i.e. a ~600pt drag downward from the very
1934
+ // top of a web page, which is pull-to-refresh: it reloads, and a reload
1935
+ // clears every field the sweep is about to fill. A peer called `sweep "all"`
1936
+ // on a half-filled form "a live grenade" for exactly this, and their run
1937
+ // logged `after 4 up to reach the top`. The claim in the comment above this
1938
+ // loop — that the top is "reached and left alone rather than pulled past" —
1939
+ // was describing an intention the code did not implement.
1940
+ //
1941
+ // So the confirming gesture is a deliberate 60pt nudge instead. It still
1942
+ // moves a page with anywhere left to go, which is the entire job, and it does
1943
+ // not overscroll far enough to refresh when there is not.
1944
+ let upSteps = 0;
1945
+ let upStalls = 0;
1946
+ let atTop = step.from === 'here';
1947
+ if (!atTop) {
1948
+ let last = await sectionHere(deviceQuery, options);
1949
+ for (let i = 0; i < limit; i += 1) {
1950
+ await scrollOne(deviceQuery, udid, 'up', ctx, upStalls ? { spanPt: TOP_CONFIRM_PT } : undefined);
1951
+ const now = await sectionHere(deviceQuery, options);
1952
+ upSteps += 1;
1953
+ // Measured, not inferred from labels: an `up` that moves nothing means we
1954
+ // are at the top.
1955
+ if (scrollDelta(last, now).moved === false) {
1956
+ upStalls += 1;
1957
+ if (upStalls >= 2) { atTop = true; break; }
1958
+ } else {
1959
+ upStalls = 0;
1960
+ }
1961
+ last = now;
1962
+ }
1963
+ }
1964
+
1965
+ const seen = new Map();
1966
+ const filled = [];
1967
+ const sections = [];
1968
+ const travelled = [];
1969
+ let atBottom = false;
1970
+ let section = 0;
1971
+
1972
+ // One read per section, and the bottom is detected at the *start* of the next
1973
+ // iteration rather than at the end of this one. That ordering is what makes
1974
+ // every screenful — including the last — get merged and filled: an earlier
1975
+ // version measured movement after scrolling and broke before reading, which
1976
+ // silently discarded the final section.
1977
+ let prev = null;
1978
+ let stalls = 0;
1979
+ for (; section < limit; section += 1) {
1980
+ const here = await sectionHere(deviceQuery, options);
1981
+ if (prev) {
1982
+ const delta = scrollDelta(prev, here);
1983
+ // Two consecutive stalls, not one.
1984
+ //
1985
+ // A single stall reading is not the bottom, and acting on one is why a
1986
+ // sweep kept jumping from the top straight to the end and never reading
1987
+ // the form in between: any sticky element inside the page — a heading
1988
+ // that stays put, a floating widget — makes one median read as zero. The
1989
+ // operator's diagnosis was exactly this: *"I feel like you miss the form,
1990
+ // you either scroll to the end or to the beginning."*
1991
+ //
1992
+ // Coverage beats stopping early. A wasted section costs about a second; a
1993
+ // missed section costs the whole point of sweeping.
1994
+ if (delta.moved === false) {
1995
+ stalls += 1;
1996
+ if (stalls >= 2) { atBottom = true; break; }
1997
+ } else {
1998
+ stalls = 0;
1999
+ }
2000
+ if (delta.px != null) travelled.push(delta.px);
2001
+ }
2002
+
2003
+ const fresh = here.filter((r) => !seen.has(sweepKey(r)));
2004
+ for (const r of here) if (!seen.has(sweepKey(r))) seen.set(sweepKey(r), { ...r, section: section + 1 });
2005
+ sections.push({ section: section + 1, elements: here.length, fresh: fresh.length });
2006
+
2007
+ // Anything to fill in this section? Do it here, while it is on screen —
2008
+ // which is the whole reason this beats finding a field and then trying to
2009
+ // scroll back to it.
2010
+ if (fill) {
2011
+ for (const [label, text] of Object.entries(fill)) {
2012
+ if (!here.some((r) => alnum(r.label).includes(alnum(label)))) continue;
2013
+ try {
2014
+ await runStep(deviceQuery, udid, {
2015
+ action: step.paste === false ? 'type' : 'paste', into: label, text: String(text),
2016
+ }, ctx);
2017
+ filled.push(`${JSON.stringify(label)} in section ${section + 1}`);
2018
+ } catch (err) {
2019
+ filled.push(`${JSON.stringify(label)} FAILED in section ${section + 1}: ${err.message.split('\n')[0].slice(0, 90)}`);
2020
+ }
2021
+ delete fill[label];
2022
+ }
2023
+ }
2024
+
2025
+ if (wanted && [...seen.values()].some((r) => alnum(r.label).includes(alnum(wanted)))) break;
2026
+ if (fill && !Object.keys(fill).length) break;
2027
+ prev = here;
2028
+ await scrollOne(deviceQuery, udid, 'down', ctx);
2029
+ }
2030
+
2031
+ const all = [...seen.values()];
2032
+ ctx.sweep = all;
2033
+ const hits = wanted ? all.filter((r) => alnum(r.label).includes(alnum(wanted))) : [];
2034
+ const listed = (wanted ? hits : all).slice(0, 30)
2035
+ .map((r) => `[${r.section}] ${JSON.stringify(String(r.label).slice(0, 36))} @${r.x},${r.y}`);
2036
+ const unfilled = fill ? Object.keys(fill) : [];
2037
+ return `swept ${sections.length} section(s)`
2038
+ + `${upSteps ? ` after ${upSteps} up to reach the ${atTop ? 'top' : 'start'}` : ''}`
2039
+ + `${atBottom ? ', reached the bottom' : ', budget spent before the bottom'}`
2040
+ + `${travelled.length ? ` (each gesture moved ${travelled.map((t) => Math.abs(t)).join(', ')}pt)` : ''};`
2041
+ + ` ${all.length} distinct element(s)`
2042
+ + (filled.length ? `; filled ${filled.join(', ')}` : '')
2043
+ + (unfilled.length ? `; NOT FOUND anywhere: ${unfilled.map((u) => JSON.stringify(u)).join(', ')}` : '')
2044
+ + (wanted ? `; ${hits.length} match ${JSON.stringify(wanted)}` : '')
2045
+ + (listed.length ? `: ${listed.join(', ')}` : '');
2046
+ }
2047
+
597
2048
  async function runStep(deviceQuery, udid, step, ctx) {
598
2049
  switch (step.action) {
599
2050
  case 'tap': {
@@ -601,7 +2052,8 @@ async function runStep(deviceQuery, udid, step, ctx) {
601
2052
  // Screen memory first: a familiar screen needs no tree read and no OCR.
602
2053
  const found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh });
603
2054
  await input.tapPoint(udid, found.target.x, found.target.y, { durationMs: step.durationMs });
604
- return `tapped "${found.target.label}" at ${found.target.x},${found.target.y} (${found.from}${found.from === 'memory' ? ` d=${found.distance}` : ''}, via ${found.target.source})`;
2055
+ return `tapped "${found.target.label}" at ${found.target.x},${found.target.y} (${found.from}${found.from === 'memory' ? ` d=${found.distance}` : ''}, via ${found.target.source})`
2056
+ + relabelledNote(found);
605
2057
  }
606
2058
  case 'tapAt': {
607
2059
  const geo = await ctx.screen();
@@ -618,14 +2070,62 @@ async function runStep(deviceQuery, udid, step, ctx) {
618
2070
  await input.tapPoint(udid, x, y, { durationMs: step.durationMs });
619
2071
  return `tapped ${x},${y}`;
620
2072
  }
2073
+ case 'clear': {
2074
+ // Empty a field. Item 61's oldest half: re-typing appends, and there was
2075
+ // no way to empty anything — not in simframe, and not in XCUITest, Appium
2076
+ // or idb either, none of which has a clear primitive. Command-A then
2077
+ // Delete over HID is the standard answer, and it is layout-independent
2078
+ // because all three are key positions rather than glyphs.
2079
+ const field = await focusField(deviceQuery, udid, { ...step, into: goalOf(step) }, ctx);
2080
+ await input.clearField(udid);
2081
+ // Verified the same way a write is, and with the same rule: OCR may
2082
+ // confirm, never deny. An empty field reads as absence of text, which is
2083
+ // exactly what a sensor that cannot see the field also reports — so only
2084
+ // the tree's own empty-valued control counts as proof it worked.
2085
+ const seen = await fieldContents(deviceQuery, field.found.target, '', ctx);
2086
+ const emptied = seen && seen.landed === false && String(seen.value) === '';
2087
+ return `cleared ${field.where}${emptied ? ' = ""' : ' [unconfirmed — nothing reads back this field\'s contents]'}`;
2088
+ }
621
2089
  case 'type': {
622
2090
  if (step.into) {
623
2091
  const field = await focusField(deviceQuery, udid, step, ctx);
624
- await input.typeText(udid, step.text ?? step.value);
625
- return `typed into ${field.where}${field.quiet}${field.waited}`;
2092
+ // Replace rather than append, when asked. The field is already focused,
2093
+ // so this costs two key events and no extra resolution.
2094
+ if (step.clear) await input.clearField(udid);
2095
+ // With `into` present, `value` is the *selector*, not the text.
2096
+ //
2097
+ // Reported as `typed into "Asset*" … = "Asset*"` — the field's own
2098
+ // label read back as its contents, because `step.text ?? step.value`
2099
+ // fell through to the selector when no text was given. That is not a
2100
+ // reporting quirk: it means the selector was typed into the field.
2101
+ const sent = textToSend(step);
2102
+ await input.typeText(udid, sent);
2103
+ let back = readbackNote(sent, await fieldContents(deviceQuery, field.found.target, sent, ctx));
2104
+ // Verifying the outcome instead of the precondition puts the race where
2105
+ // it actually shows up. If the keystrokes arrived before the field had
2106
+ // focus they are simply gone — one local retry costs about 200ms, and
2107
+ // throwing here cost an aborted batch and a model round trip.
2108
+ if (back.empty) {
2109
+ await input.tapPoint(udid, field.found.target.x, field.found.target.y);
2110
+ await input.typeText(udid, sent);
2111
+ back = readbackNote(sent, await fieldContents(deviceQuery, field.found.target, sent, ctx));
2112
+ if (back.empty) {
2113
+ throw new Error(
2114
+ `typed into ${field.where} twice and the field still reads empty — the text is not landing.`
2115
+ + ' paste is more reliable than type on this path; keys sends literal characters, not named keys.',
2116
+ );
2117
+ }
2118
+ journalWrite(udid, step, sent, back, ctx);
2119
+ return `typed into ${field.where}${back.note} [took two attempts; the first keystrokes did not land]`;
2120
+ }
2121
+ journalWrite(udid, step, sent, back, ctx);
2122
+ return `typed into ${field.where}${back.note}${back.landed ? '' : field.quiet}`;
626
2123
  }
627
2124
  await input.typeText(udid, step.text ?? step.value);
628
- return 'typed text';
2125
+ // No selector, so there is nothing to read back — which is a fine trade
2126
+ // for typing into whatever Safari's own form chevrons focused, but it
2127
+ // must not be reported as though the text was seen to land.
2128
+ return 'typed text [unconfirmed — no field named, so nothing was read back]';
629
2129
  }
630
2130
  case 'paste': {
631
2131
  // Long strings are much faster on the pasteboard than through the
@@ -634,11 +2134,22 @@ async function runStep(deviceQuery, udid, step, ctx) {
634
2134
  // report success anyway.
635
2135
  if (step.into) {
636
2136
  const field = await focusField(deviceQuery, udid, step, ctx);
637
- await input.pasteText(udid, step.text ?? step.value);
638
- return `pasted into ${field.where}${field.quiet}${field.waited}`;
2137
+ const sent = textToSend(step);
2138
+ if (step.clear) await input.clearField(udid);
2139
+ await input.pasteText(udid, sent);
2140
+ const seen = await fieldContents(deviceQuery, field.found.target, sent, ctx);
2141
+ const back = readbackNote(sent, seen);
2142
+ if (back.empty) {
2143
+ throw new Error(
2144
+ `pasted into ${field.where} and the field reads empty — the text did not land.`
2145
+ + ' A first paste can raise the system paste-consent dialog and lose the text; dismiss it and retry.',
2146
+ );
2147
+ }
2148
+ journalWrite(udid, step, sent, back, ctx);
2149
+ return `pasted into ${field.where}${back.note}${back.landed ? '' : field.quiet}`;
639
2150
  }
640
2151
  await input.pasteText(udid, step.text ?? step.value);
641
- return 'pasted into the focused field';
2152
+ return 'pasted into the focused field [unconfirmed — no field named, so nothing was read back]';
642
2153
  }
643
2154
  case 'swipe': {
644
2155
  const from = { x: step.from?.[0] ?? step.from?.x, y: step.from?.[1] ?? step.from?.y };
@@ -695,6 +2206,10 @@ async function runStep(deviceQuery, udid, step, ctx) {
695
2206
  const r = await intent.chooseAny(udid, { prefer: step.value ?? step.prefer, geo });
696
2207
  return `chose "${r.label}" of ${r.optionCount} options`;
697
2208
  }
2209
+ case 'sweep':
2210
+ return sweep(deviceQuery, udid, step, ctx);
2211
+ case 'seek':
2212
+ return seek(deviceQuery, udid, step, ctx);
698
2213
  case 'settle': {
699
2214
  const w = await api.waitFor(deviceQuery, {
700
2215
  mode: step.mode ?? 'stable',
@@ -747,18 +2262,82 @@ async function runStep(deviceQuery, udid, step, ctx) {
747
2262
  // about it.
748
2263
  case 'scrollTo': {
749
2264
  const query = step.value ?? step.target ?? step.label;
750
- const dir = String(step.direction ?? 'down').toLowerCase();
2265
+ // Which way, and it now reads the answer instead of assuming it.
2266
+ //
2267
+ // Reported three times in one round and the most expensive single defect
2268
+ // across five runs: the target sat at **y = −693**, above the viewport,
2269
+ // and this scrolled *down* six times — moving away on every iteration,
2270
+ // with the offset printed in its own error each time — then advised the
2271
+ // tool it already is. The operator watching called it "scrolled too much
2272
+ // and trying to scroll more like a loop", which is exactly what it was.
2273
+ //
2274
+ // The tree knows where the element is whenever it is in the tree at all,
2275
+ // so ask before each gesture and follow the sign. An explicit
2276
+ // `direction` still wins, for a caller who knows better.
2277
+ const asked = step.direction ? String(step.direction).toLowerCase() : null;
751
2278
  const max = Math.min(MAX_SCROLLS, step.maxScrolls ?? 6);
2279
+ let dir = asked ?? 'down';
2280
+ let reversed = false;
2281
+ // Where the target is, when the tree knows. `null` means no evidence, and
2282
+ // that distinction is load bearing: guessing "up" without it scrolls to
2283
+ // the top of a web page, which **triggers pull-to-refresh**, reloads the
2284
+ // page and changes the screen hash — defeating the end-detection below
2285
+ // and reading, from outside, as an endless loop. Observed live.
2286
+ const offsetSays = async () => {
2287
+ if (asked) return asked;
2288
+ try {
2289
+ const { entry, points } = await api.readScreenWith(deviceQuery, { useOcr: false, options: ctx.options });
2290
+ const hit = matching.resolve(entry.targets ?? [], String(query));
2291
+ const y = hit?.target?.y;
2292
+ if (!Number.isFinite(y)) return null;
2293
+ if (y < 0) return 'up';
2294
+ if (y > (points?.height ?? Infinity)) return 'down';
2295
+ return dir;
2296
+ } catch {
2297
+ return null;
2298
+ }
2299
+ };
752
2300
  for (let i = 0; i <= max; i += 1) {
753
2301
  try {
754
2302
  const found = await api.locate(deviceQuery, query, { index: step.index, refresh: i > 0 });
755
2303
  return `"${found.target.label}" is in view at ${found.target.x},${found.target.y}` +
756
- (i ? ` after ${i} scroll${i === 1 ? '' : 's'}` : ' already');
2304
+ (i ? ` after ${i} scroll${i === 1 ? '' : 's'} ${dir}` : ' already');
757
2305
  } catch (err) {
758
- if (i === max) throw new Error(`scrolled ${dir} ${max}x without finding ${query}: ${err.message}`);
2306
+ if (i === max) {
2307
+ throw new Error(`scrolled ${dir} ${max}x without finding ${query}: ${err.message}`);
2308
+ }
759
2309
  }
2310
+ const evidence = await offsetSays();
2311
+ if (evidence) dir = evidence;
2312
+ const wasAt = await hashNow(deviceQuery, ctx.options);
760
2313
  await runStep(deviceQuery, udid, { action: 'scroll', value: dir }, ctx);
761
- await api.waitFor(deviceQuery, { mode: 'stable', stableMs: 250, timeoutMs: 2500, options: ctx.options });
2314
+ // A scroll either moves immediately or not at all, so it does not need a
2315
+ // transition's budget. Six iterations at 2,500ms was most of why this
2316
+ // read as a loop from outside — *"it looks like a loop"* — rather than
2317
+ // as a search.
2318
+ await api.waitFor(deviceQuery, { mode: 'stable', stableMs: 200, timeoutMs: SCROLL_SETTLE_MS, options: ctx.options });
2319
+ // A scroll that moved nothing means we are against an end. Burning the
2320
+ // rest of the budget against it is what the operator watched happen:
2321
+ // *"your scroll still looks unstable, you're just scrolling past the
2322
+ // page"*. Reverse once — the target may be behind us, and on a page
2323
+ // whose fields never enter the tree there is no offset to follow — then
2324
+ // stop rather than thrash.
2325
+ const nowAt = await hashNow(deviceQuery, ctx.options);
2326
+ if (wasAt && nowAt && wasAt === nowAt) {
2327
+ // Reverse only on evidence. Without it we do not know the target is
2328
+ // behind us, and scrolling blindly the other way is how a web page
2329
+ // gets pulled to refresh.
2330
+ if (reversed || !evidence) {
2331
+ throw new Error(
2332
+ `${query} is not reachable by scrolling: ${dir} stopped moving after ${i + 1} attempt(s)`
2333
+ + (evidence ? ' and so did the other way.' : ' and the tree does not say where it is,'
2334
+ + ' so there is no direction to try.')
2335
+ + ' It may not be in the accessibility tree at all — read the screen, or aim at a coordinate.',
2336
+ );
2337
+ }
2338
+ reversed = true;
2339
+ dir = dir === 'down' ? 'up' : 'down';
2340
+ }
762
2341
  }
763
2342
  throw new Error(`could not bring ${query} into view`);
764
2343
  }
@@ -766,9 +2345,45 @@ async function runStep(deviceQuery, udid, step, ctx) {
766
2345
  // Wait for a selector rather than a label, so it works on screens the
767
2346
  // accessibility tree never described.
768
2347
  case 'waitFor': {
2348
+ // One string, or any of several.
2349
+ //
2350
+ // Reported: a wait on *"any login or dashboard content"* spent **120
2351
+ // seconds** while the login screen was already there — and the failure
2352
+ // message itself listed `Email`, `Password`, `Remember me`. A phrase like
2353
+ // that is a disjunction, and resolving it as one intent asks the matcher
2354
+ // for something no single element answers.
2355
+ //
2356
+ // `{"waitFor": {"any": ["Email", "Dashboard"]}}` says it directly, and
2357
+ // the first to appear wins. Which is also the honest division of labour:
2358
+ // simframe resolves an intent to an element, and *which of several
2359
+ // outcomes am I waiting for* is the caller's question to phrase.
2360
+ const alternatives = Array.isArray(step.any) ? step.any.filter(Boolean).map(String) : null;
769
2361
  const query = step.value ?? step.target ?? step.text;
2362
+ if (!alternatives && !query) throw new Error('usage: {"waitFor": "text"} or {"waitFor": {"any": ["a", "b"]}}');
770
2363
  const limit = Date.now() + (step.timeoutMs ?? 8000);
771
2364
  let lastError = 'never appeared';
2365
+ if (alternatives) {
2366
+ for (let attempt = 0; ; attempt += 1) {
2367
+ for (const one of alternatives) {
2368
+ try {
2369
+ const found = await api.locate(deviceQuery, one, { refresh: attempt > 0 });
2370
+ return `${JSON.stringify(one)} appeared at ${found.target.x},${found.target.y}`
2371
+ + ` (first of ${alternatives.length} awaited)`;
2372
+ } catch (err) {
2373
+ lastError = err.message;
2374
+ }
2375
+ }
2376
+ if (Date.now() >= limit) {
2377
+ throw new Error(
2378
+ `none of ${alternatives.length} awaited strings appeared`
2379
+ + ` (${alternatives.map((a) => JSON.stringify(a)).join(', ')}) in ${step.timeoutMs ?? 8000}ms.`
2380
+ + ` Last: ${lastError}`,
2381
+ );
2382
+ }
2383
+ await api.waitFor(deviceQuery, { mode: 'stable', stableMs: 200, timeoutMs: 700, options: ctx.options })
2384
+ .catch(() => null);
2385
+ }
2386
+ }
772
2387
  for (let attempt = 0; ; attempt += 1) {
773
2388
  try {
774
2389
  const found = await api.locate(deviceQuery, query, { index: step.index, refresh: attempt > 0 });
@@ -806,11 +2421,44 @@ async function runStep(deviceQuery, udid, step, ctx) {
806
2421
  const want = String(step.is ?? (step.gone ? 'gone' : 'visible')).toLowerCase();
807
2422
  let found = null;
808
2423
  try {
809
- found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh });
2424
+ // Fresh by default, and this was the single most expensive finding in a
2425
+ // reported round. `assert REVIEW is enabled` failed while the map
2426
+ // printed by that very call showed the button enabled three lines
2427
+ // below: the assert had resolved against screen *memory*, and state
2428
+ // changes without a screen's identity changing.
2429
+ //
2430
+ // A verification step reading a cached map is self-defeating. It costs
2431
+ // one perception pass, which is the same trade Phase 11.5 made for the
2432
+ // trailing map and for the same reason — a read is cheaper than the
2433
+ // round trip a wrong verdict causes.
2434
+ //
2435
+ // The reporter's framing is why this is worth a paragraph: *"I put an
2436
+ // assert in to be careful, and being careful is what broke the flow.
2437
+ // The lesson an agent learns is 'do not assert inside batches', which
2438
+ // is the opposite of what you want learned."*
2439
+ found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh !== false, options: ctx.options });
810
2440
  } catch (err) {
811
2441
  if (want === 'gone') return `${query} is gone`;
812
2442
  throw new Error(`${query}: ${err.message}`);
813
2443
  }
2444
+ // A state that disagrees earns one more look, because the state may have
2445
+ // arrived between the action and the assert — which is exactly what a
2446
+ // batch does. `tap` already re-takes a stale baseline and says so.
2447
+ const disagrees = (t) => (want === 'enabled' && t.enabled === false)
2448
+ || (want === 'disabled' && t.enabled !== false);
2449
+ let retook = '';
2450
+ if (disagrees(found.target)) {
2451
+ await api.waitFor(deviceQuery, {
2452
+ mode: 'settle', stableMs: 250, timeoutMs: ASSERT_RETAKE_MS, options: ctx.options,
2453
+ }).catch(() => null);
2454
+ try {
2455
+ const again = await api.locate(deviceQuery, query, { index: step.index, refresh: true, options: ctx.options });
2456
+ if (!disagrees(again.target)) {
2457
+ found = again;
2458
+ retook = ' [state arrived between the action and the assert; re-read from the live screen]';
2459
+ }
2460
+ } catch { /* keep the first reading and report it */ }
2461
+ }
814
2462
  const t = found.target;
815
2463
  switch (want) {
816
2464
  case 'visible':
@@ -818,11 +2466,11 @@ async function runStep(deviceQuery, udid, step, ctx) {
818
2466
  case 'gone':
819
2467
  throw new Error(`${query} is still on screen at ${t.x},${t.y}`);
820
2468
  case 'enabled':
821
- if (t.enabled === false) throw new Error(`"${t.label}" is disabled`);
822
- return `"${t.label}" is enabled`;
2469
+ if (t.enabled === false) throw new Error(`"${t.label}" is disabled, and still disabled on a second look`);
2470
+ return `"${t.label}" is enabled${retook}`;
823
2471
  case 'disabled':
824
- if (t.enabled !== false) throw new Error(`"${t.label}" is not disabled`);
825
- return `"${t.label}" is disabled`;
2472
+ if (t.enabled !== false) throw new Error(`"${t.label}" is not disabled, on two looks`);
2473
+ return `"${t.label}" is disabled${retook}`;
826
2474
  case 'value': {
827
2475
  const expected = String(step.equals ?? step.text ?? '');
828
2476
  const actual = [t.label, t.value, ...(t.aliases ?? [])].filter(Boolean).join(' ');