simframe 0.8.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/actions.js CHANGED
@@ -6,6 +6,8 @@ import * as api from './index.js';
6
6
  import * as graph from './graph.js';
7
7
  import * as input from './input.js';
8
8
  import * as intent from './intent.js';
9
+ import * as metrics from './metrics.js';
10
+ import * as screenmap from './screenmap.js';
9
11
  import { launchApp, openUrl, setPermission, terminateApp } from './platform/index.js';
10
12
 
11
13
  const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
@@ -25,11 +27,22 @@ const MAX_PAUSE_MS = 5000;
25
27
  * falls out at `reaction` instead. That bounds the cost of the honest case
26
28
  * rather than the broken one.
27
29
  */
30
+ /**
31
+ * The stillness window stays fixed, and that is a decision rather than an
32
+ * oversight. Learning a stillness window is the half of Phase 11 that was
33
+ * reverted for cause: a wait that ends early never observes the pauses that
34
+ * come later, so the estimator ratchets itself down and the graph learns
35
+ * transitions that never happened. See docs/BENCHMARKS.md, Phase 11.
36
+ */
28
37
  const FOCUS_STABLE_MS = 250;
29
38
  /**
30
- * Long enough for a slow capture loop to produce a frame or two. The screenshot
31
- * engine idles at 1.5 fps — 667 ms between frames — so anything under that is a
32
- * verdict reached before there was anything to look at.
39
+ * The cold defaults, unchanged, for a field this screen has not been measured
40
+ * focusing. `graph.focusPlan` takes over once it has been, and may only make
41
+ * the wait longer.
42
+ *
43
+ * 900 ms is long enough for a slow capture loop to produce a frame or two. The
44
+ * screenshot engine idles at 1.5 fps — 667 ms between frames — so anything
45
+ * under that is a verdict reached before there was anything to look at.
33
46
  */
34
47
  const FOCUS_REACTION_MS = 900;
35
48
  const FOCUS_TIMEOUT_MS = 3000;
@@ -112,6 +125,12 @@ export async function runScript(
112
125
  // Rebuild the HID session and retry once when a hardware button provably
113
126
  // did nothing. Off only for a caller deliberately testing that path.
114
127
  recoverInput = true,
128
+ // What this run is called and how few steps it could take, for the flow
129
+ // record. A bare `sim_do` has neither and says so with nulls rather than
130
+ // inventing a name — an unnamed run still gets timed, it just cannot be
131
+ // compared against a human baseline.
132
+ flowName = null,
133
+ minSteps = null,
115
134
  options,
116
135
  } = {},
117
136
  ) {
@@ -119,6 +138,24 @@ export async function runScript(
119
138
  const { device } = await api.ensureDaemon(deviceQuery, options);
120
139
  const udid = device.udid;
121
140
  const startedAt = Date.now();
141
+ // Measurement only. Nothing below reads these, and a failure to write one
142
+ // can never change what a step does — see `note`.
143
+ const flowId = metrics.newFlowId();
144
+ const escalations = [];
145
+ const verdicts = [];
146
+ // Named noteEscalation, not note: the step loop below declares its own
147
+ // `note` string for the no-visible-change suffix, which shadowed this and
148
+ // turned every escalating verdict into a failed step reading "note is not a
149
+ // function". The try/catch inside here could not help — the throw was at the
150
+ // call site, one scope out. Instrumentation that can fail a flow is worse
151
+ // than no instrumentation.
152
+ const noteEscalation = (record) => {
153
+ try {
154
+ escalations.push(metrics.recordEscalation(udid, { flowId, flowName, ...record }));
155
+ } catch {
156
+ /* instrumentation must not be able to fail a flow it is only watching */
157
+ }
158
+ };
122
159
 
123
160
  const needsInput = steps.some((s) => ACTION_STEPS.has(normalizeStep(s).action));
124
161
  if (needsInput) {
@@ -145,10 +182,35 @@ export async function runScript(
145
182
  // moves nothing and is not a failure, so an unbounded retry would rebuild the
146
183
  // session and press again on every such step for no reason.
147
184
  let inputRecovered = false;
185
+ /**
186
+ * The previous step's transition, still to be measured.
187
+ *
188
+ * Its pause profile cannot be read while the step is running — that is the
189
+ * biased measurement that corrupted the graph — so it is read one step later,
190
+ * off a frame history whose end nothing about the wait decided. See
191
+ * `api.longestQuietGap`.
192
+ */
193
+ let pendingGap = null;
194
+ const measurePendingGap = async () => {
195
+ if (!pendingGap) return;
196
+ const { from, step: prevStep, actionAt } = pendingGap;
197
+ pendingGap = null;
198
+ try {
199
+ const history = (await api.getState(deviceQuery, { options })).state.history ?? [];
200
+ const trueGapMs = api.longestQuietGap(history, actionAt);
201
+ if (trueGapMs != null) graph.noteTrueGap(udid, from, prevStep, trueGapMs);
202
+ } catch {
203
+ /* a statistic nothing acts on must never be able to fail a flow */
204
+ }
205
+ };
148
206
 
149
207
  for (const [i, raw] of steps.entries()) {
150
208
  const step = normalizeStep(raw);
151
209
  const stepStart = Date.now();
210
+ // Before anything else, and before this step disturbs the screen: the
211
+ // previous transition is definitely over by now, so its true pause profile
212
+ // is readable.
213
+ await measurePendingGap();
152
214
  // The baseline for "did the screen react" must predate the action itself.
153
215
  const beforeState = (await api.getState(deviceQuery, { options })).state;
154
216
  const before = beforeState.hash;
@@ -164,14 +226,72 @@ export async function runScript(
164
226
  // What this action did last time it was taken here, if ever.
165
227
  const prediction = verify && beforeScreen?.hash ? graph.predict(udid, beforeScreen, step) : null;
166
228
  try {
167
- let detail = await runStep(deviceQuery, udid, step, { screen, options, frames });
229
+ // How long this transition has cost before, on this screen, for this
230
+ // action. A cold edge gets the old fixed default and says so; a measured
231
+ // one gets p95 plus a margin. Research §7.
232
+ //
233
+ // Read before the step, not after, because one of the waits it informs
234
+ // happens *inside* the step: a `type into` taps the field and waits for
235
+ // focus before it types, and that wait used to be three constants.
236
+ const learned = verify && beforeScreen?.hash ? graph.timingFor(udid, beforeScreen, step) : null;
237
+ const focus = {
238
+ plan: graph.focusPlan(learned, {
239
+ reactionMs: FOCUS_REACTION_MS,
240
+ timeoutMs: FOCUS_TIMEOUT_MS,
241
+ stillnessMs: FOCUS_STABLE_MS,
242
+ keyboardUp: Boolean(beforeScreen?.keyboard),
243
+ }),
244
+ observedMs: null,
245
+ };
246
+ let detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
247
+ // How long this screen must hold still before it counts as settled.
248
+ //
249
+ // 500 ms was a constant paid by every step of every flow, and it is the
250
+ // reducible half of a settle: the rest is the transition genuinely
251
+ // taking time. An edge whose transition has never paused mid-flight
252
+ // needs 150 ms of quiet, not 500. Capped at the caller's own value, so
253
+ // this can only ever shorten a wait.
254
+ // NOT YET USED TO DECIDE ANYTHING, and the reason is worth the space.
255
+ //
256
+ // `graph.stillnessFor` computes a shorter window from the longest pause
257
+ // ever observed inside this transition, and measured live it made the
258
+ // Settings flow 8.0 s instead of 11.5 s — and wrong. Eight runs in a row
259
+ // failed at step 2 with the screen still showing Settings root, because
260
+ // step 1's settle returned mid-push, `screenIdentity` then read the
261
+ // screen we had not left yet, and the graph learned root -> root as a
262
+ // verified edge and started predicting it.
263
+ //
264
+ // The flaw is in the estimator, not the idea: the gap statistic is
265
+ // gathered only from what a wait itself observed, so a wait that ends
266
+ // early never sees the pauses that come later, the gaps look like zero,
267
+ // the window ratchets down, and the next wait ends earlier still. A
268
+ // self-reinforcing bias with a wrong graph at the end of it.
269
+ //
270
+ // The unbiased estimator is available and is a separate piece of work:
271
+ // the frame history holds every frame's timestamp and diff, so the true
272
+ // motion profile of a transition can be computed *after* it is over
273
+ // rather than from inside the wait that cut it short. Until then the
274
+ // gaps are recorded and not acted on — measuring is safe, and this is
275
+ // Phase 11's own rule that a learned number may only ever shorten a
276
+ // wait, applied to itself.
277
+ const stillness = step.stableMs ?? stableMs;
278
+ const stillnessPlan = learned ? graph.stillnessFor(learned, stillness) : { cold: true };
279
+ // A settle is not satisfied until the screen has held still for
280
+ // `stillness`, so a budget below that can never be met — and the learned
281
+ // p95 is measured from waits that include the stillness window, which
282
+ // makes it self-consistent but not self-evidently so. A tab switch with
283
+ // a p95 of 90ms would get a 240ms budget and then time out at 240ms
284
+ // waiting for 500ms of quiet, turning every fast edge into a failure.
285
+ const floorMs = stillness + 250;
286
+ const budgetMs = step.timeoutMs
287
+ ?? (learned && !learned.cold ? Math.max(learned.timeoutMs, floorMs) : timeoutMs);
168
288
  const settleFor = async () => {
169
289
  if (!autoSettle || !ACTION_STEPS.has(step.action)) return null;
170
290
  const w = await api.waitFor(deviceQuery, {
171
291
  mode: 'settle',
172
292
  since: before,
173
- stableMs: step.stableMs ?? stableMs,
174
- timeoutMs: step.timeoutMs ?? timeoutMs,
293
+ stableMs: stillness,
294
+ timeoutMs: budgetMs,
175
295
  options,
176
296
  });
177
297
  return {
@@ -180,9 +300,64 @@ export async function runScript(
180
300
  sawChange: w.sawChange,
181
301
  stalled: Boolean(w.stalled),
182
302
  noVisibleChange: Boolean(w.noVisibleChange),
303
+ // What this wait was allowed, and where the number came from. A
304
+ // timeout nobody can explain is how a fixed sleep comes back as a
305
+ // constant with a comment.
306
+ budgetMs,
307
+ stillnessMs: stillness,
308
+ quietGapMs: w.quietGapMs,
309
+ // The baseline had already finished moving when the wait began, so it
310
+ // was re-taken from the live screen. Surfaced because it means the
311
+ // step before this one had not finished when this one started.
312
+ staleBaseline: Boolean(w.staleBaseline),
313
+ // Something moved in one region only — a switch, a radio dot, a
314
+ // segment highlight. Worth saying, because it is the difference
315
+ // between "the action did nothing" and "the action did something the
316
+ // whole-screen mean cannot see".
317
+ smallChange: Boolean(w.smallChange),
318
+ timing: learned
319
+ ? {
320
+ p50: learned.p50,
321
+ p95: learned.p95,
322
+ samples: learned.samples,
323
+ cold: learned.cold,
324
+ gapP95: learned.gapP95,
325
+ gapSamples: learned.gapSamples,
326
+ // What it *would* have been, for the eval that has to happen
327
+ // before this is trusted with a wait.
328
+ stillnessWouldBe: stillnessPlan.stillnessMs ?? null,
329
+ }
330
+ : null,
183
331
  };
184
332
  };
185
333
  let settled = await settleFor();
334
+ // A screen that is still working earns more time; a screen doing nothing
335
+ // visible has already answered. Research §7: keep waiting past p95 only
336
+ // while the transition classifier says something is loading, and never
337
+ // past Nielsen's 10 s — at which point it escalates with the timing
338
+ // attached rather than waiting longer.
339
+ if (settled && !settled.ok && !settled.noVisibleChange && learned && !learned.cold) {
340
+ const kind = (await api.getState(deviceQuery, { options })).state.transition?.kind;
341
+ const verdict = graph.slowerThanUsual({
342
+ elapsedMs: settled.waitedMs, p95: learned.p95, settled: false, kind,
343
+ });
344
+ if (verdict.keepWaiting) {
345
+ const remaining = graph.HARD_CAP_MS - settled.waitedMs;
346
+ const more = await api.waitFor(deviceQuery, {
347
+ mode: 'settle', since: before, stableMs: stillness, timeoutMs: remaining, options,
348
+ });
349
+ settled = {
350
+ ...settled,
351
+ ok: more.satisfied,
352
+ waitedMs: settled.waitedMs + more.waitedMs,
353
+ sawChange: settled.sawChange || more.sawChange,
354
+ quietGapMs: Math.max(settled.quietGapMs ?? 0, more.quietGapMs ?? 0),
355
+ slowerThanUsual: verdict.note,
356
+ };
357
+ } else if (verdict.slower) {
358
+ settled = { ...settled, slowerThanUsual: verdict.note };
359
+ }
360
+ }
186
361
 
187
362
  // A hardware button that moved nothing did not arrive.
188
363
  //
@@ -226,14 +401,66 @@ export async function runScript(
226
401
  // for. Requiring both meant a screen that settled slowly recorded
227
402
  // nothing at all.
228
403
  endScreen = afterScreen;
229
- if (afterScreen.confirmed && afterScreen.hash) {
230
- graph.record(udid, { from: beforeScreen, action: step, to: afterScreen, kind });
404
+ // An action with no observed effect teaches the graph nothing, and
405
+ // recording it teaches something false.
406
+ //
407
+ // This is the second half of the same bug. A settle that returned on a
408
+ // stale baseline reported `ok` for a tap that moved nothing, and the
409
+ // recorder asked only whether the *reading* was confirmed — so
410
+ // `root -> root` went in as a verified edge and started being
411
+ // predicted. Re-baselining stops the settle lying; this stops the
412
+ // graph learning from a step that has no evidence behind it either way.
413
+ //
414
+ // It does cost a real case for now: a control that genuinely returns to
415
+ // the same screen — a toggle — is invisible to the change detector at
416
+ // eight times below its threshold, so it reads as no-visible-change and
417
+ // its edge is no longer recorded. That is the right trade while the
418
+ // detector cannot see it, and it comes back on its own once it can.
419
+ const noEvidence = Boolean(settled?.noVisibleChange);
420
+ if (afterScreen.confirmed && afterScreen.hash && !noEvidence) {
421
+ // The observed cost of this transition, which is what makes the next
422
+ // one adaptive. Only from a settle that was actually satisfied: a
423
+ // timeout is not a measurement of how long the screen takes, it is a
424
+ // measurement of how long we were prepared to wait.
425
+ graph.record(udid, {
426
+ from: beforeScreen, action: step, to: afterScreen, kind,
427
+ settleMs: settled?.ok ? settled.waitedMs : undefined,
428
+ // The pause statistic is worth having from any settle that saw the
429
+ // screen move, satisfied or not: a transition that paused for
430
+ // 400ms and then timed out is exactly the case a 150ms stillness
431
+ // window would have got wrong.
432
+ quietGapMs: settled?.sawChange ? settled.quietGapMs : undefined,
433
+ // The focus wait's own distribution, kept apart from the step's.
434
+ // Only set when a field was tapped and visibly took focus.
435
+ focusMs: focus.observedMs ?? undefined,
436
+ });
437
+ pendingGap = { from: beforeScreen, step, actionAt: stepStart };
438
+ carriedScreen = afterScreen;
439
+ } else if (afterScreen.confirmed && afterScreen.hash) {
440
+ // Where we are is still known; only what got us here is not worth
441
+ // remembering. Carrying it saves the next step a perception pass.
231
442
  carriedScreen = afterScreen;
232
443
  }
233
444
  }
234
445
 
235
446
  const wrongTurn = wrongTurnFrom(verification);
236
- const note = settled?.noVisibleChange ? ' [no visible change]' : '';
447
+ // `[no visible change]` after a launch is ambiguous between two very
448
+ // different things, and a real session read it the wrong way twice:
449
+ // "the app was already in front, so nothing needed to move" and "the app
450
+ // did not come forward". Measured on this Xcode, `simctl launch` *does*
451
+ // front an already-running app, so the first reading is the likely one —
452
+ // but likely is not the same as said, and the step is the only place that
453
+ // can say it.
454
+ const launchNote = step.action === 'launch' && settled?.noVisibleChange
455
+ ? ' [the screen did not change, so this app was already in front — or it did not come forward]'
456
+ : '';
457
+ const note = launchNote
458
+ + (settled?.smallChange ? ' [a small change, in one region only]' : '')
459
+ + (settled?.noVisibleChange ? ' [no visible change]' : '')
460
+ + (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
461
+ + (settled?.blackFrames
462
+ ? ` [${settled.blackFrames} black frame(s) waited through${settled.blackMs ? `, still black after ${settled.blackMs}ms` : ''}]`
463
+ : '');
237
464
  results.push({
238
465
  index: i,
239
466
  action: step.action,
@@ -244,6 +471,25 @@ export async function runScript(
244
471
  settled,
245
472
  });
246
473
  const halt = haltDecision({ verification, stopOnUnexpected, continueOnError });
474
+ if (verification?.verdict) verdicts.push(verification.verdict);
475
+ if (metrics.ESCALATING_VERDICTS.has(verification?.verdict)) {
476
+ noteEscalation({
477
+ stepIndex: i,
478
+ fingerprint: beforeScreen?.hash ?? null,
479
+ reason: 'verification_failed',
480
+ candidates: [],
481
+ // A halted run is a decision simframe made and stopped on; a step
482
+ // that moved nothing carries on and leaves the judgement to whoever
483
+ // reads the result.
484
+ outcome: halt.halt ? 'failed' : 'escalated_to_model',
485
+ wallMs: Date.now() - stepStart,
486
+ detail: `${verification.verdict}: ${verification.detail}`
487
+ + (settled?.slowerThanUsual ? ` [${settled.slowerThanUsual}]` : '')
488
+ + (settled?.timing && !settled.timing.cold
489
+ ? ` [waited ${settled.waitedMs}ms of a ${settled.budgetMs}ms budget; p95 ${settled.timing.p95}ms]`
490
+ : ''),
491
+ });
492
+ }
247
493
  if (halt.halt) {
248
494
  results[results.length - 1].ok = false;
249
495
  results[results.length - 1].error = halt.error;
@@ -252,20 +498,56 @@ export async function runScript(
252
498
  }
253
499
  } catch (err) {
254
500
  results.push({ index: i, action: step.action, ok: false, ms: Date.now() - stepStart, error: err.message });
501
+ const why = metrics.reasonForStepError(step, err);
502
+ noteEscalation({
503
+ stepIndex: i,
504
+ fingerprint: beforeScreen?.hash ?? metrics.fingerprintNow(udid, screenmap),
505
+ reason: why.reason,
506
+ candidates: why.candidates,
507
+ tried: why.tried,
508
+ outcome: 'failed',
509
+ wallMs: Date.now() - stepStart,
510
+ detail: err.message,
511
+ });
255
512
  failed = true;
256
513
  if (!continueOnError) break;
257
514
  }
258
515
  }
259
516
 
517
+ // The last step has no next step to measure it, and its transition is over by
518
+ // the time the loop exits.
519
+ await measurePendingGap();
520
+
521
+ const wallMs = Date.now() - startedAt;
522
+ try {
523
+ metrics.recordFlow(udid, metrics.flowRecordFrom({
524
+ flowId,
525
+ flowName,
526
+ udid,
527
+ startedAt,
528
+ wallMs,
529
+ stepsTaken: results.length,
530
+ totalSteps: steps.length,
531
+ minSteps,
532
+ imagesSent: frames.length,
533
+ escalations,
534
+ verdicts,
535
+ completed: !failed && results.length === steps.length,
536
+ }));
537
+ } catch {
538
+ /* as above: a flow that ran is not a flow that failed because of a log */
539
+ }
540
+
260
541
  return {
261
542
  device,
262
543
  // Returned so a run that verified end to end can be handed straight to
263
544
  // navigate.saveFlow without the caller reassembling what it just ran.
264
545
  steps,
546
+ flowId,
265
547
  endScreen,
266
548
  results,
267
549
  ok: !failed,
268
- totalMs: Date.now() - startedAt,
550
+ totalMs: wallMs,
269
551
  ranSteps: results.length,
270
552
  totalSteps: steps.length,
271
553
  frames,
@@ -284,18 +566,31 @@ export async function runScript(
284
566
  */
285
567
  async function focusField(deviceQuery, udid, step, ctx) {
286
568
  const found = await api.locate(deviceQuery, step.into, { index: step.index, refresh: step.refresh });
569
+ // What this field has cost to focus before, on this screen. Cold, or with no
570
+ // verification running, that is exactly the three constants above; measured,
571
+ // it can only be longer. `graph.focusPlan` carries the reason it is either.
572
+ const plan = ctx.focus?.plan ?? {
573
+ reactionMs: FOCUS_REACTION_MS, timeoutMs: FOCUS_TIMEOUT_MS, cold: true, from: 'no timing in hand',
574
+ };
575
+ const tappedAt = Date.now();
287
576
  await input.tapPoint(udid, found.target.x, found.target.y);
288
577
  const focused = await api.waitFor(deviceQuery, {
289
578
  mode: 'settle',
290
579
  stableMs: FOCUS_STABLE_MS,
291
- reactionMs: FOCUS_REACTION_MS,
292
- timeoutMs: FOCUS_TIMEOUT_MS,
580
+ reactionMs: plan.reactionMs,
581
+ timeoutMs: plan.timeoutMs,
293
582
  options: ctx.options,
294
583
  });
584
+ // Only a wait that was satisfied is a measurement of how long focus takes. A
585
+ // reaction window that ran out measures how long we were prepared to watch a
586
+ // screen that did not move, and banking that would teach the edge the cost of
587
+ // its own impatience — the estimator mistake learned stillness made.
588
+ if (ctx.focus && focused.satisfied) ctx.focus.observedMs = Date.now() - tappedAt;
295
589
  return {
296
590
  found,
297
591
  where: `"${found.target.label}" at ${found.target.x},${found.target.y}`,
298
592
  quiet: focused.satisfied ? '' : ' [the field did not visibly take focus]',
593
+ waited: focused.satisfied && !plan.cold ? ` [focus in ${focused.waitedMs}ms, ${plan.from}]` : '',
299
594
  };
300
595
  }
301
596
 
@@ -327,7 +622,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
327
622
  if (step.into) {
328
623
  const field = await focusField(deviceQuery, udid, step, ctx);
329
624
  await input.typeText(udid, step.text ?? step.value);
330
- return `typed into ${field.where}${field.quiet}`;
625
+ return `typed into ${field.where}${field.quiet}${field.waited}`;
331
626
  }
332
627
  await input.typeText(udid, step.text ?? step.value);
333
628
  return 'typed text';
@@ -340,7 +635,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
340
635
  if (step.into) {
341
636
  const field = await focusField(deviceQuery, udid, step, ctx);
342
637
  await input.pasteText(udid, step.text ?? step.value);
343
- return `pasted into ${field.where}${field.quiet}`;
638
+ return `pasted into ${field.where}${field.quiet}${field.waited}`;
344
639
  }
345
640
  await input.pasteText(udid, step.text ?? step.value);
346
641
  return 'pasted into the focused field';
@@ -420,6 +715,14 @@ async function runStep(deviceQuery, udid, step, ctx) {
420
715
  return `"${node.label ?? target}" appeared`;
421
716
  } catch (err) {
422
717
  lastError = err.message;
718
+ // Same rule as `waitFor`, and `matchElement` says it in its own
719
+ // words: a query that matched several elements has found them all
720
+ // already.
721
+ if (/matched \d+ elements/.test(err.message)) {
722
+ throw new Error(
723
+ `${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
724
+ );
725
+ }
423
726
  }
424
727
  await sleep(250);
425
728
  }
@@ -472,6 +775,22 @@ async function runStep(deviceQuery, udid, step, ctx) {
472
775
  return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`;
473
776
  } catch (err) {
474
777
  lastError = err.message;
778
+ // Waiting cannot make a thing unique.
779
+ //
780
+ // Reported from a real session: a wait on an ambiguous string spent
781
+ // the full 30 s and then listed four matches, all four of which were
782
+ // on the very first frame. The disambiguation is good and it arrived
783
+ // twenty-nine seconds after everything it needed. `ambiguous` means
784
+ // the target is *present*, several times over — which is precisely
785
+ // the case where more time changes nothing.
786
+ //
787
+ // Distinguished by the tag at the throw site rather than by reading
788
+ // the message, because "not on this screen" tags the same reason.
789
+ if (metrics.escalationOf(err)?.ambiguous) {
790
+ throw new Error(
791
+ `${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
792
+ );
793
+ }
475
794
  }
476
795
  if (Date.now() >= limit) break;
477
796
  await sleep(POLL_MS);
package/src/analyze.js CHANGED
@@ -35,6 +35,42 @@ export function signatureDiff(a, b) {
35
35
  return sum / a.length / 255;
36
36
  }
37
37
 
38
+ /**
39
+ * A change small enough that the whole-screen mean cannot see it.
40
+ *
41
+ * `changed` in both daemons is `signatureDiff > 0.004`, a mean over a 4x8 grid
42
+ * of gray means. Measured on this device, an iOS switch flipping:
43
+ *
44
+ * mean diff 0.001348 — a third of the threshold, so: not a change
45
+ * max cell delta 0.043137 — one cell of thirty-two, in row 1
46
+ *
47
+ * So the entire class of small binary controls — switches, radio dots,
48
+ * checkboxes, segment highlights — changes nothing as far as the daemon is
49
+ * concerned, and a step that flips one reports `no-visible-change`, which is a
50
+ * verdict that escalates.
51
+ *
52
+ * The threshold sits in a measured gap rather than being chosen. Eighty seconds
53
+ * of a *static* screen gave a largest per-cell delta of 0.003922, in row 0,
54
+ * which is the status-bar clock ticking over — the only thing moving. So the
55
+ * separation is 0.0039 against 0.0431, eleven times, and 0.012 is three times
56
+ * the noise and three and a half times under the signal. No row is excluded:
57
+ * the clock does not reach the threshold, which is a better reason to ignore it
58
+ * than a structural exclusion that would also blind the nav bar.
59
+ *
60
+ * What this deliberately does **not** do is feed stillness. `stableForMs` stays
61
+ * on the mean, because a blinking text caret is a small localised change and a
62
+ * screen with a cursor in it would otherwise never settle. The two signals are
63
+ * independent by design: this one answers "did the action do anything", and the
64
+ * mean answers "has the screen finished moving".
65
+ */
66
+ export const CELL_CHANGE = 0.012;
67
+
68
+ /** The largest single-region change between two signatures, 0-1. */
69
+ export function maxCellDelta(a, b) {
70
+ const deltas = regionDeltas(a, b);
71
+ return deltas.length ? Math.max(...deltas) : 0;
72
+ }
73
+
38
74
  /** Per-region change fractions, so callers can tell a toast from a screen push. */
39
75
  export function regionDeltas(a, b) {
40
76
  if (!a || !b || a.length !== b.length) return a ? a.map(() => 1) : [];
@@ -69,6 +105,40 @@ export function signatureToHex(sig) {
69
105
  return sig.map((v) => v.toString(16).padStart(2, '0')).join('');
70
106
  }
71
107
 
108
+ /**
109
+ * Is this frame black — not dark, black?
110
+ *
111
+ * The capture wedge (docs/BENCHMARKS.md) leaves `simctl io screenshot`
112
+ * succeeding and returning 0 non-black pixels of 3,162,132, and simframe's own
113
+ * frames go the same way: the display pipeline stops rendering while
114
+ * everything about the capture path keeps reporting success. It self-recovers
115
+ * most times and a device restart cures the rest, so the useful thing is to
116
+ * notice early — before a settle spends its whole budget deciding a black
117
+ * screen is a calm one.
118
+ *
119
+ * The signature is already computed for every frame, so this costs 32 integer
120
+ * comparisons and no decode. A real screen does not come close: measured on
121
+ * this device's Settings root, the 32 bytes ran 191–245.
122
+ *
123
+ * The threshold is a level, not a fraction, and it is on the *maximum*: one
124
+ * cell with anything in it is enough to say the display is rendering. That
125
+ * matters because a dark-mode screen, a video, or a splash on black are all
126
+ * legitimately near-zero in most cells and this must not call them faults.
127
+ *
128
+ * And it says "the frames are black", never "the simulator is wedged". A
129
+ * screen can be black because the app drew black. What makes it a wedge is
130
+ * that it stays black while input is being delivered, and only the caller
131
+ * knows that.
132
+ */
133
+ export const BLACK_LEVEL = 8;
134
+
135
+ export function isBlackFrame(sig, { level = BLACK_LEVEL } = {}) {
136
+ const bytes = typeof sig === 'string' ? hexToSignature(sig) : sig;
137
+ if (!bytes?.length) return false;
138
+ for (const b of bytes) if (b > level) return false;
139
+ return true;
140
+ }
141
+
72
142
  export function hexToSignature(hex) {
73
143
  const out = [];
74
144
  for (let i = 0; i < hex.length; i += 2) out.push(parseInt(hex.slice(i, i + 2), 16));