simframe 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/actions.js CHANGED
@@ -27,11 +27,22 @@ const MAX_PAUSE_MS = 5000;
27
27
  * falls out at `reaction` instead. That bounds the cost of the honest case
28
28
  * rather than the broken one.
29
29
  */
30
+ /**
31
+ * The stillness window stays fixed, and that is a decision rather than an
32
+ * oversight. Learning a stillness window is the half of Phase 11 that was
33
+ * reverted for cause: a wait that ends early never observes the pauses that
34
+ * come later, so the estimator ratchets itself down and the graph learns
35
+ * transitions that never happened. See docs/BENCHMARKS.md, Phase 11.
36
+ */
30
37
  const FOCUS_STABLE_MS = 250;
31
38
  /**
32
- * Long enough for a slow capture loop to produce a frame or two. The screenshot
33
- * engine idles at 1.5 fps — 667 ms between frames — so anything under that is a
34
- * verdict reached before there was anything to look at.
39
+ * The cold defaults, unchanged, for a field this screen has not been measured
40
+ * focusing. `graph.focusPlan` takes over once it has been, and may only make
41
+ * the wait longer.
42
+ *
43
+ * 900 ms is long enough for a slow capture loop to produce a frame or two. The
44
+ * screenshot engine idles at 1.5 fps — 667 ms between frames — so anything
45
+ * under that is a verdict reached before there was anything to look at.
35
46
  */
36
47
  const FOCUS_REACTION_MS = 900;
37
48
  const FOCUS_TIMEOUT_MS = 3000;
@@ -140,7 +151,7 @@ export async function runScript(
140
151
  // than no instrumentation.
141
152
  const noteEscalation = (record) => {
142
153
  try {
143
- escalations.push(metrics.recordEscalation(udid, { flowId, ...record }));
154
+ escalations.push(metrics.recordEscalation(udid, { flowId, flowName, ...record }));
144
155
  } catch {
145
156
  /* instrumentation must not be able to fail a flow it is only watching */
146
157
  }
@@ -171,10 +182,35 @@ export async function runScript(
171
182
  // moves nothing and is not a failure, so an unbounded retry would rebuild the
172
183
  // session and press again on every such step for no reason.
173
184
  let inputRecovered = false;
185
+ /**
186
+ * The previous step's transition, still to be measured.
187
+ *
188
+ * Its pause profile cannot be read while the step is running — that is the
189
+ * biased measurement that corrupted the graph — so it is read one step later,
190
+ * off a frame history whose end nothing about the wait decided. See
191
+ * `api.longestQuietGap`.
192
+ */
193
+ let pendingGap = null;
194
+ const measurePendingGap = async () => {
195
+ if (!pendingGap) return;
196
+ const { from, step: prevStep, actionAt } = pendingGap;
197
+ pendingGap = null;
198
+ try {
199
+ const history = (await api.getState(deviceQuery, { options })).state.history ?? [];
200
+ const trueGapMs = api.longestQuietGap(history, actionAt);
201
+ if (trueGapMs != null) graph.noteTrueGap(udid, from, prevStep, trueGapMs);
202
+ } catch {
203
+ /* a statistic nothing acts on must never be able to fail a flow */
204
+ }
205
+ };
174
206
 
175
207
  for (const [i, raw] of steps.entries()) {
176
208
  const step = normalizeStep(raw);
177
209
  const stepStart = Date.now();
210
+ // Before anything else, and before this step disturbs the screen: the
211
+ // previous transition is definitely over by now, so its true pause profile
212
+ // is readable.
213
+ await measurePendingGap();
178
214
  // The baseline for "did the screen react" must predate the action itself.
179
215
  const beforeState = (await api.getState(deviceQuery, { options })).state;
180
216
  const before = beforeState.hash;
@@ -190,11 +226,24 @@ export async function runScript(
190
226
  // What this action did last time it was taken here, if ever.
191
227
  const prediction = verify && beforeScreen?.hash ? graph.predict(udid, beforeScreen, step) : null;
192
228
  try {
193
- let detail = await runStep(deviceQuery, udid, step, { screen, options, frames });
194
229
  // How long this transition has cost before, on this screen, for this
195
230
  // action. A cold edge gets the old fixed default and says so; a measured
196
231
  // one gets p95 plus a margin. Research §7.
232
+ //
233
+ // Read before the step, not after, because one of the waits it informs
234
+ // happens *inside* the step: a `type into` taps the field and waits for
235
+ // focus before it types, and that wait used to be three constants.
197
236
  const learned = verify && beforeScreen?.hash ? graph.timingFor(udid, beforeScreen, step) : null;
237
+ const focus = {
238
+ plan: graph.focusPlan(learned, {
239
+ reactionMs: FOCUS_REACTION_MS,
240
+ timeoutMs: FOCUS_TIMEOUT_MS,
241
+ stillnessMs: FOCUS_STABLE_MS,
242
+ keyboardUp: Boolean(beforeScreen?.keyboard),
243
+ }),
244
+ observedMs: null,
245
+ };
246
+ let detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
198
247
  // How long this screen must hold still before it counts as settled.
199
248
  //
200
249
  // 500 ms was a constant paid by every step of every flow, and it is the
@@ -257,6 +306,15 @@ export async function runScript(
257
306
  budgetMs,
258
307
  stillnessMs: stillness,
259
308
  quietGapMs: w.quietGapMs,
309
+ // The baseline had already finished moving when the wait began, so it
310
+ // was re-taken from the live screen. Surfaced because it means the
311
+ // step before this one had not finished when this one started.
312
+ staleBaseline: Boolean(w.staleBaseline),
313
+ // Something moved in one region only — a switch, a radio dot, a
314
+ // segment highlight. Worth saying, because it is the difference
315
+ // between "the action did nothing" and "the action did something the
316
+ // whole-screen mean cannot see".
317
+ smallChange: Boolean(w.smallChange),
260
318
  timing: learned
261
319
  ? {
262
320
  p50: learned.p50,
@@ -343,7 +401,23 @@ export async function runScript(
343
401
  // for. Requiring both meant a screen that settled slowly recorded
344
402
  // nothing at all.
345
403
  endScreen = afterScreen;
346
- if (afterScreen.confirmed && afterScreen.hash) {
404
+ // An action with no observed effect teaches the graph nothing, and
405
+ // recording it teaches something false.
406
+ //
407
+ // This is the second half of the same bug. A settle that returned on a
408
+ // stale baseline reported `ok` for a tap that moved nothing, and the
409
+ // recorder asked only whether the *reading* was confirmed — so
410
+ // `root -> root` went in as a verified edge and started being
411
+ // predicted. Re-baselining stops the settle lying; this stops the
412
+ // graph learning from a step that has no evidence behind it either way.
413
+ //
414
+ // It does cost a real case for now: a control that genuinely returns to
415
+ // the same screen — a toggle — is invisible to the change detector at
416
+ // eight times below its threshold, so it reads as no-visible-change and
417
+ // its edge is no longer recorded. That is the right trade while the
418
+ // detector cannot see it, and it comes back on its own once it can.
419
+ const noEvidence = Boolean(settled?.noVisibleChange);
420
+ if (afterScreen.confirmed && afterScreen.hash && !noEvidence) {
347
421
  // The observed cost of this transition, which is what makes the next
348
422
  // one adaptive. Only from a settle that was actually satisfied: a
349
423
  // timeout is not a measurement of how long the screen takes, it is a
@@ -356,13 +430,37 @@ export async function runScript(
356
430
  // 400ms and then timed out is exactly the case a 150ms stillness
357
431
  // window would have got wrong.
358
432
  quietGapMs: settled?.sawChange ? settled.quietGapMs : undefined,
433
+ // The focus wait's own distribution, kept apart from the step's.
434
+ // Only set when a field was tapped and visibly took focus.
435
+ focusMs: focus.observedMs ?? undefined,
359
436
  });
437
+ pendingGap = { from: beforeScreen, step, actionAt: stepStart };
438
+ carriedScreen = afterScreen;
439
+ } else if (afterScreen.confirmed && afterScreen.hash) {
440
+ // Where we are is still known; only what got us here is not worth
441
+ // remembering. Carrying it saves the next step a perception pass.
360
442
  carriedScreen = afterScreen;
361
443
  }
362
444
  }
363
445
 
364
446
  const wrongTurn = wrongTurnFrom(verification);
365
- const note = settled?.noVisibleChange ? ' [no visible change]' : '';
447
+ // `[no visible change]` after a launch is ambiguous between two very
448
+ // different things, and a real session read it the wrong way twice:
449
+ // "the app was already in front, so nothing needed to move" and "the app
450
+ // did not come forward". Measured on this Xcode, `simctl launch` *does*
451
+ // front an already-running app, so the first reading is the likely one —
452
+ // but likely is not the same as said, and the step is the only place that
453
+ // can say it.
454
+ const launchNote = step.action === 'launch' && settled?.noVisibleChange
455
+ ? ' [the screen did not change, so this app was already in front — or it did not come forward]'
456
+ : '';
457
+ const note = launchNote
458
+ + (settled?.smallChange ? ' [a small change, in one region only]' : '')
459
+ + (settled?.noVisibleChange ? ' [no visible change]' : '')
460
+ + (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
461
+ + (settled?.blackFrames
462
+ ? ` [${settled.blackFrames} black frame(s) waited through${settled.blackMs ? `, still black after ${settled.blackMs}ms` : ''}]`
463
+ : '');
366
464
  results.push({
367
465
  index: i,
368
466
  action: step.action,
@@ -416,6 +514,10 @@ export async function runScript(
416
514
  }
417
515
  }
418
516
 
517
+ // The last step has no next step to measure it, and its transition is over by
518
+ // the time the loop exits.
519
+ await measurePendingGap();
520
+
419
521
  const wallMs = Date.now() - startedAt;
420
522
  try {
421
523
  metrics.recordFlow(udid, metrics.flowRecordFrom({
@@ -464,18 +566,31 @@ export async function runScript(
464
566
  */
465
567
  async function focusField(deviceQuery, udid, step, ctx) {
466
568
  const found = await api.locate(deviceQuery, step.into, { index: step.index, refresh: step.refresh });
569
+ // What this field has cost to focus before, on this screen. Cold, or with no
570
+ // verification running, that is exactly the three constants above; measured,
571
+ // it can only be longer. `graph.focusPlan` carries the reason it is either.
572
+ const plan = ctx.focus?.plan ?? {
573
+ reactionMs: FOCUS_REACTION_MS, timeoutMs: FOCUS_TIMEOUT_MS, cold: true, from: 'no timing in hand',
574
+ };
575
+ const tappedAt = Date.now();
467
576
  await input.tapPoint(udid, found.target.x, found.target.y);
468
577
  const focused = await api.waitFor(deviceQuery, {
469
578
  mode: 'settle',
470
579
  stableMs: FOCUS_STABLE_MS,
471
- reactionMs: FOCUS_REACTION_MS,
472
- timeoutMs: FOCUS_TIMEOUT_MS,
580
+ reactionMs: plan.reactionMs,
581
+ timeoutMs: plan.timeoutMs,
473
582
  options: ctx.options,
474
583
  });
584
+ // Only a wait that was satisfied is a measurement of how long focus takes. A
585
+ // reaction window that ran out measures how long we were prepared to watch a
586
+ // screen that did not move, and banking that would teach the edge the cost of
587
+ // its own impatience — the estimator mistake learned stillness made.
588
+ if (ctx.focus && focused.satisfied) ctx.focus.observedMs = Date.now() - tappedAt;
475
589
  return {
476
590
  found,
477
591
  where: `"${found.target.label}" at ${found.target.x},${found.target.y}`,
478
592
  quiet: focused.satisfied ? '' : ' [the field did not visibly take focus]',
593
+ waited: focused.satisfied && !plan.cold ? ` [focus in ${focused.waitedMs}ms, ${plan.from}]` : '',
479
594
  };
480
595
  }
481
596
 
@@ -507,7 +622,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
507
622
  if (step.into) {
508
623
  const field = await focusField(deviceQuery, udid, step, ctx);
509
624
  await input.typeText(udid, step.text ?? step.value);
510
- return `typed into ${field.where}${field.quiet}`;
625
+ return `typed into ${field.where}${field.quiet}${field.waited}`;
511
626
  }
512
627
  await input.typeText(udid, step.text ?? step.value);
513
628
  return 'typed text';
@@ -520,7 +635,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
520
635
  if (step.into) {
521
636
  const field = await focusField(deviceQuery, udid, step, ctx);
522
637
  await input.pasteText(udid, step.text ?? step.value);
523
- return `pasted into ${field.where}${field.quiet}`;
638
+ return `pasted into ${field.where}${field.quiet}${field.waited}`;
524
639
  }
525
640
  await input.pasteText(udid, step.text ?? step.value);
526
641
  return 'pasted into the focused field';
@@ -600,6 +715,14 @@ async function runStep(deviceQuery, udid, step, ctx) {
600
715
  return `"${node.label ?? target}" appeared`;
601
716
  } catch (err) {
602
717
  lastError = err.message;
718
+ // Same rule as `waitFor`, and `matchElement` says it in its own
719
+ // words: a query that matched several elements has found them all
720
+ // already.
721
+ if (/matched \d+ elements/.test(err.message)) {
722
+ throw new Error(
723
+ `${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
724
+ );
725
+ }
603
726
  }
604
727
  await sleep(250);
605
728
  }
@@ -652,6 +775,22 @@ async function runStep(deviceQuery, udid, step, ctx) {
652
775
  return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`;
653
776
  } catch (err) {
654
777
  lastError = err.message;
778
+ // Waiting cannot make a thing unique.
779
+ //
780
+ // Reported from a real session: a wait on an ambiguous string spent
781
+ // the full 30 s and then listed four matches, all four of which were
782
+ // on the very first frame. The disambiguation is good and it arrived
783
+ // twenty-nine seconds after everything it needed. `ambiguous` means
784
+ // the target is *present*, several times over — which is precisely
785
+ // the case where more time changes nothing.
786
+ //
787
+ // Distinguished by the tag at the throw site rather than by reading
788
+ // the message, because "not on this screen" tags the same reason.
789
+ if (metrics.escalationOf(err)?.ambiguous) {
790
+ throw new Error(
791
+ `${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
792
+ );
793
+ }
655
794
  }
656
795
  if (Date.now() >= limit) break;
657
796
  await sleep(POLL_MS);
package/src/analyze.js CHANGED
@@ -35,6 +35,42 @@ export function signatureDiff(a, b) {
35
35
  return sum / a.length / 255;
36
36
  }
37
37
 
38
+ /**
39
+ * A change small enough that the whole-screen mean cannot see it.
40
+ *
41
+ * `changed` in both daemons is `signatureDiff > 0.004`, a mean over a 4x8 grid
42
+ * of gray means. Measured on this device, an iOS switch flipping:
43
+ *
44
+ * mean diff 0.001348 — a third of the threshold, so: not a change
45
+ * max cell delta 0.043137 — one cell of thirty-two, in row 1
46
+ *
47
+ * So the entire class of small binary controls — switches, radio dots,
48
+ * checkboxes, segment highlights — changes nothing as far as the daemon is
49
+ * concerned, and a step that flips one reports `no-visible-change`, which is a
50
+ * verdict that escalates.
51
+ *
52
+ * The threshold sits in a measured gap rather than being chosen. Eighty seconds
53
+ * of a *static* screen gave a largest per-cell delta of 0.003922, in row 0,
54
+ * which is the status-bar clock ticking over — the only thing moving. So the
55
+ * separation is 0.0039 against 0.0431, eleven times, and 0.012 is three times
56
+ * the noise and three and a half times under the signal. No row is excluded:
57
+ * the clock does not reach the threshold, which is a better reason to ignore it
58
+ * than a structural exclusion that would also blind the nav bar.
59
+ *
60
+ * What this deliberately does **not** do is feed stillness. `stableForMs` stays
61
+ * on the mean, because a blinking text caret is a small localised change and a
62
+ * screen with a cursor in it would otherwise never settle. The two signals are
63
+ * independent by design: this one answers "did the action do anything", and the
64
+ * mean answers "has the screen finished moving".
65
+ */
66
+ export const CELL_CHANGE = 0.012;
67
+
68
+ /** The largest single-region change between two signatures, 0-1. */
69
+ export function maxCellDelta(a, b) {
70
+ const deltas = regionDeltas(a, b);
71
+ return deltas.length ? Math.max(...deltas) : 0;
72
+ }
73
+
38
74
  /** Per-region change fractions, so callers can tell a toast from a screen push. */
39
75
  export function regionDeltas(a, b) {
40
76
  if (!a || !b || a.length !== b.length) return a ? a.map(() => 1) : [];
@@ -69,6 +105,40 @@ export function signatureToHex(sig) {
69
105
  return sig.map((v) => v.toString(16).padStart(2, '0')).join('');
70
106
  }
71
107
 
108
+ /**
109
+ * Is this frame black — not dark, black?
110
+ *
111
+ * The capture wedge (docs/BENCHMARKS.md) leaves `simctl io screenshot`
112
+ * succeeding and returning 0 non-black pixels of 3,162,132, and simframe's own
113
+ * frames go the same way: the display pipeline stops rendering while
114
+ * everything about the capture path keeps reporting success. It self-recovers
115
+ * most times and a device restart cures the rest, so the useful thing is to
116
+ * notice early — before a settle spends its whole budget deciding a black
117
+ * screen is a calm one.
118
+ *
119
+ * The signature is already computed for every frame, so this costs 32 integer
120
+ * comparisons and no decode. A real screen does not come close: measured on
121
+ * this device's Settings root, the 32 bytes ran 191–245.
122
+ *
123
+ * The threshold is a level, not a fraction, and it is on the *maximum*: one
124
+ * cell with anything in it is enough to say the display is rendering. That
125
+ * matters because a dark-mode screen, a video, or a splash on black are all
126
+ * legitimately near-zero in most cells and this must not call them faults.
127
+ *
128
+ * And it says "the frames are black", never "the simulator is wedged". A
129
+ * screen can be black because the app drew black. What makes it a wedge is
130
+ * that it stays black while input is being delivered, and only the caller
131
+ * knows that.
132
+ */
133
+ export const BLACK_LEVEL = 8;
134
+
135
+ export function isBlackFrame(sig, { level = BLACK_LEVEL } = {}) {
136
+ const bytes = typeof sig === 'string' ? hexToSignature(sig) : sig;
137
+ if (!bytes?.length) return false;
138
+ for (const b of bytes) if (b > level) return false;
139
+ return true;
140
+ }
141
+
72
142
  export function hexToSignature(hex) {
73
143
  const out = [];
74
144
  for (let i = 0; i < hex.length; i += 2) out.push(parseInt(hex.slice(i, i + 2), 16));
package/src/cli.js CHANGED
@@ -5,6 +5,7 @@ import path from 'node:path';
5
5
  import { runDaemon, DEFAULTS } from './daemon.js';
6
6
  import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
7
7
  import * as actions from './actions.js';
8
+ import * as analyze from './analyze.js';
8
9
  import * as api from './index.js';
9
10
  import * as input from './input.js';
10
11
  import * as baseline from './baseline.js';
@@ -20,6 +21,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
20
21
  simframe start [device] start the capture loop in the background
21
22
  simframe stop [device|--all] stop the capture loop
22
23
  simframe status [device] show daemon and newest-frame status
24
+ simframe input reset rebuild the HID session (see doctor)
23
25
  simframe frame [device] write the newest frame to a file
24
26
  simframe state [device] print frame metadata and the change map
25
27
  simframe mark [device] print the current frame hash, to use as --since
@@ -310,6 +312,29 @@ async function main() {
310
312
  return;
311
313
  }
312
314
 
315
+ // Rebuild the daemon's HID session, and nothing else.
316
+ //
317
+ // The narrow remedy for the narrow fault. Restarting the daemon also cures
318
+ // a stale session and throws away the frame ring and every warm cache to do
319
+ // it, which is the difference between a fix and a power cycle.
320
+ case 'input': {
321
+ const what = positional[0];
322
+ if (what !== 'reset') throw new Error('usage: simframe input reset [--device <udid>]');
323
+ const dev = await resolveDevice(device);
324
+ const before = await input.sessionHealth(dev.udid);
325
+ const reset = await input.resetSession(dev.udid);
326
+ if (!reset) {
327
+ // Said plainly rather than as a success: there is no daemon holding a
328
+ // session to rebuild, so nothing was wrong and nothing was done.
329
+ console.log(`no simframed session to rebuild for ${dev.name} — the daemon is not running, or this device is not driven by it`);
330
+ process.exitCode = 1;
331
+ return;
332
+ }
333
+ console.log(`rebuilt the HID session for ${dev.name}`
334
+ + (before.stale ? `\n it was stale: ${before.reason}` : '\n it did not report stale; rebuilt anyway, as asked'));
335
+ return;
336
+ }
337
+
313
338
  case 'status': {
314
339
  const udids = device
315
340
  ? [(await resolveDevice(device)).udid]
@@ -951,23 +976,52 @@ async function main() {
951
976
  case 'escalations': {
952
977
  const dev = await resolveDevice(flags.device);
953
978
  const records = metrics.readEscalations(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
954
- const b = metrics.breakdown(records);
979
+ const b = metrics.breakdown(records, {
980
+ session: flags.session === true ? metrics.sessionId() : (flags.session ? String(flags.session) : null),
981
+ flow: flags.flow ? String(flags.flow) : null,
982
+ });
955
983
  if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify(b, null, 2)}\n`);
956
984
  emit(flags, b, [
957
985
  `${b.total} escalation${b.total === 1 ? '' : 's'} on ${dev.name}`,
958
986
  ...metrics.REASONS
959
987
  .filter((r) => b.by_reason[r])
960
988
  .sort((a, c) => b.by_reason[c] - b.by_reason[a])
961
- .map((r) => ` ${r.padEnd(20)} ${String(b.by_reason[r]).padStart(4)} would be removed by: ${metrics.FACULTY[r]}`),
989
+ .map((r) => ` ${r.padEnd(20)} ${String(b.by_reason[r]).padStart(4)} `
990
+ + (metrics.BUILT_FACULTIES.has(metrics.FACULTY[r])
991
+ ? `not removed by: ${metrics.FACULTY[r]} [built]`
992
+ : `would be removed by: ${metrics.FACULTY[r]}`)),
962
993
  b.total ? '' : null,
963
994
  b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
964
995
  // Said out loud rather than left for someone to discover: the rate is
965
996
  // 1.0 while no faculty exists, so the breakdown above is the number
966
997
  // that decides the next phase.
967
998
  b.total && b.avoidable_escalation_rate === 1
968
- ? ' every reason maps to a faculty that is not built yet, so this rate is 1.0 by construction. The per-reason counts are the steering wheel.'
999
+ ? (metrics.REASONS.some((r) => b.by_reason[r] && metrics.BUILT_FACULTIES.has(metrics.FACULTY[r]))
1000
+ ? ' the rate is 1.0 because nothing resolves locally yet. A reason marked [built] is not a queue waiting on a phase — it is evidence the phase that shipped is not sufficient.'
1001
+ : ' every reason maps to a faculty that is not built yet, so this rate is 1.0 by construction. The per-reason counts are the steering wheel.')
969
1002
  : null,
970
1003
  b.total ? `model turns spent on escalations: ${b.model_turns_spent}` : null,
1004
+ // The log is per-device and shared. Said before the counts are used,
1005
+ // not after: two agents on one booted simulator write one interleaved
1006
+ // file, and CLAUDE.md makes these counts the thing that picks the next
1007
+ // faculty. A pooled breakdown errs toward whichever session made more
1008
+ // mistakes, which is a different question.
1009
+ b.pooled
1010
+ ? 'WARNING these counts may pool more than one agent\'s work: '
1011
+ + [
1012
+ b.session_count > 1 ? `${b.session_count} sessions` : null,
1013
+ b.unattributed ? `${b.unattributed} record(s) written before sessions were logged` : null,
1014
+ ].filter(Boolean).join(', ')
1015
+ + '. Narrow with --session (this process), --session=<id>, or --flow=<name>.'
1016
+ : null,
1017
+ b.session_count > 1 ? 'sessions:' : null,
1018
+ ...(b.session_count > 1
1019
+ ? b.sessions.map((x) => ` ${x.session_id.padEnd(22)} ${String(x.count).padStart(4)} ${x.client}`)
1020
+ : []),
1021
+ Object.keys(b.by_flow).length > 1 ? 'flows:' : null,
1022
+ ...(Object.keys(b.by_flow).length > 1
1023
+ ? Object.entries(b.by_flow).slice(0, 10).map(([n, c]) => ` ${n.padEnd(28)} ${String(c).padStart(4)}`)
1024
+ : []),
971
1025
  b.top_screens.length ? 'top screens:' : null,
972
1026
  ...b.top_screens.map((s) => ` ${s.fingerprint.slice(0, 16).padEnd(18)} ${s.count}`),
973
1027
  metrics.writeError() ? `WARNING a log write failed: ${metrics.writeError()}` : null,
@@ -1228,7 +1282,18 @@ async function doctor({ json = false, strict = false, device } = {}) {
1228
1282
  // dispatched successfully and moved nothing, five runs in a row.
1229
1283
  const session = await input.sessionHealth(d.udid);
1230
1284
  if (session.stale) {
1231
- add(`input session (${d.name})`, 'warn', `stale — ${session.reason}. The next action rebuilds it automatically; simframe stop && simframe start does it now`,
1285
+ // The remedy used to read "the next action rebuilds it automatically;
1286
+ // simframe stop && simframe start does it now", and both halves were
1287
+ // wrong. The first was false wherever it mattered, because the rebuild
1288
+ // check was gated once per process and the MCP server is one process
1289
+ // for a whole session. The second names a command that fails twice:
1290
+ // `stop` needs `--device` when two simulators are booted, and then
1291
+ // refuses because a client holds the daemon, so the sequence that
1292
+ // actually works is `stop --device <udid> --force && start --device
1293
+ // <udid>` — a daemon restart, to fix a session, when rebuilding the
1294
+ // session is a thing the daemon can already do on request. It just had
1295
+ // no way in from outside. It does now.
1296
+ add(`input session (${d.name})`, 'warn', `stale — ${session.reason}. The next action rebuilds it; simframe input reset --device ${d.udid} does it now`,
1232
1297
  { key: 'input.session', value: 'stale' });
1233
1298
  } else if (caps.input.supported) {
1234
1299
  add(`input session (${d.name})`, 'ok', session.reason ?? 'current with this device session',
@@ -1252,8 +1317,20 @@ async function doctor({ json = false, strict = false, device } = {}) {
1252
1317
  if (probed.length) {
1253
1318
  const t0 = Date.now();
1254
1319
  const res = await api.getFrame(probed[0].udid);
1255
- add('capture', 'ok',
1256
- `frame #${res.state.seq} ${res.width}x${res.height} in ${Date.now() - t0}ms (age ${res.ageMs}ms)`,
1320
+ // The wedge's whole signature is that everything here reports success.
1321
+ // A black frame is 32 integer comparisons on a signature already
1322
+ // computed, and it is what separates "captured a frame" from "captured
1323
+ // a frame of a display that has stopped rendering".
1324
+ const dark = analyze.isBlackFrame(
1325
+ (res.state.history ?? []).find((h) => h.seq === res.state.seq)?.sig ?? null,
1326
+ );
1327
+ add('capture', dark ? 'warn' : 'ok',
1328
+ `frame #${res.state.seq} ${res.width}x${res.height} in ${Date.now() - t0}ms (age ${res.ageMs}ms)`
1329
+ + (dark
1330
+ ? ' — and every pixel of it is black. If the device is not showing a black screen on purpose,'
1331
+ + ' this is the display pipeline having stopped rendering; it usually recovers on its own,'
1332
+ + ` and ${'xcrun simctl shutdown'} / boot is the cure that always works.`
1333
+ : ''),
1257
1334
  { key: 'capture.frames', value: res.state.seq });
1258
1335
  // A wedged device produces the same nothing as a quiet one, so doctor has
1259
1336
  // to ask the capture loop rather than look at the frames. `fail`, not
package/src/graph.js CHANGED
@@ -348,7 +348,7 @@ export function findScreen(udid, query) {
348
348
  * 2.4 s on a screen that fetches — and because the graph is already persisted,
349
349
  * versioned and pruned.
350
350
  */
351
- function noteSettle(edge, settleMs, quietGapMs) {
351
+ function noteSettle(edge, settleMs, quietGapMs, focusMs) {
352
352
  if (Number.isFinite(settleMs) && settleMs >= 0) {
353
353
  edge.settles = [...(edge.settles ?? []), Math.round(settleMs)].slice(-TIMING_WINDOW);
354
354
  }
@@ -357,6 +357,39 @@ function noteSettle(edge, settleMs, quietGapMs) {
357
357
  if (Number.isFinite(quietGapMs) && quietGapMs >= 0) {
358
358
  edge.quietGaps = [...(edge.quietGaps ?? []), Math.round(quietGapMs)].slice(-TIMING_WINDOW);
359
359
  }
360
+ // How long the *field* took to take focus, which is a different duration from
361
+ // how long the step took: it is measured between the tap and the keyboard,
362
+ // inside a step whose settle is measured after the typing. One edge, two
363
+ // waits, so two distributions.
364
+ if (Number.isFinite(focusMs) && focusMs >= 0) {
365
+ edge.focuses = [...(edge.focuses ?? []), Math.round(focusMs)].slice(-TIMING_WINDOW);
366
+ }
367
+ }
368
+
369
+ /**
370
+ * Record the unbiased pause statistic for an edge already written.
371
+ *
372
+ * Separate from `record` because it arrives later on purpose. The biased
373
+ * `quietGaps` are gathered from inside the wait and are therefore bounded by
374
+ * when the wait chose to stop; `trueGaps` are read off the frame history once
375
+ * the transition is definitely over, which is one step later. So the edge has
376
+ * to be found again rather than passed along.
377
+ *
378
+ * Nothing reads `trueGaps` yet, and that is deliberate. Phase 11 built the
379
+ * learned stillness window on the biased statistic, it corrupted the graph
380
+ * inside an afternoon, and the lesson taken was not "use a better estimator" —
381
+ * it was that a number gets to *act* only after it has been watched for a while
382
+ * doing nothing. This is the watching.
383
+ */
384
+ export function noteTrueGap(udid, screen, step, trueGapMs) {
385
+ if (!Number.isFinite(trueGapMs) || trueGapMs < 0) return null;
386
+ const node = screen?.hash ? nearestScreen(udid, screen)?.node : null;
387
+ if (!node) return null;
388
+ const edge = node.edges?.find((e) => e.action === actionSignature(step));
389
+ if (!edge) return null;
390
+ edge.trueGaps = [...(edge.trueGaps ?? []), Math.round(trueGapMs)].slice(-TIMING_WINDOW);
391
+ save(udid, node);
392
+ return edge;
360
393
  }
361
394
 
362
395
  /**
@@ -384,16 +417,76 @@ export function stillnessFor({ gapSamples, gapP95 } = {}, fallbackMs) {
384
417
  };
385
418
  }
386
419
 
420
+ /**
421
+ * How long to wait for a tapped field to take focus.
422
+ *
423
+ * The three numbers this replaces were the last genuinely fixed waits on the
424
+ * action path: 250 ms of stillness, a 900 ms reaction window, a 3 s timeout.
425
+ * What makes them different from the step budget is the shape of the failure.
426
+ * A step budget that is too short reports `no-visible-change` and the flow can
427
+ * see it. A focus wait that is too short types into a field that does not have
428
+ * focus yet, `typeText` succeeds because input has no feedback channel, and the
429
+ * step reports that it typed — the worst shape a failure can take, and the bug
430
+ * this helper was written to fix in the first place.
431
+ *
432
+ * So this one is asymmetric on purpose: **a learned window may only lengthen
433
+ * the wait**, never shorten it. p95 of what this field has actually cost, when
434
+ * that is longer than 3 s, is a field that was being typed into too early and
435
+ * now is not. Where it is shorter, the measurement is discarded rather than
436
+ * banked as a saving — 5% of a distribution is one silent wrong type in twenty
437
+ * runs, and there is no amount of median wall time worth that.
438
+ *
439
+ * The one shortening is not a learned number at all, it is positive evidence:
440
+ * if the keyboard was already up before the tap, this tap moves a caret. There
441
+ * is no keyboard animation to wait for, so the reaction window collapses to the
442
+ * stillness window instead of paying 900 ms to watch a screen that was never
443
+ * going to move. `beforeScreen` already carries `keyboard`, so the evidence is
444
+ * free — it is the same perception pass the step was going to run anyway.
445
+ */
446
+ export function focusPlan(stats, { reactionMs, timeoutMs, stillnessMs, keyboardUp } = {}) {
447
+ if (keyboardUp) {
448
+ return {
449
+ reactionMs: stillnessMs ?? reactionMs,
450
+ timeoutMs,
451
+ cold: false,
452
+ from: 'the keyboard was already up, so this tap moves a caret',
453
+ };
454
+ }
455
+ const { focusP50, focusP95, focusSamples } = stats ?? {};
456
+ if (!Number.isFinite(focusP95) || !Number.isFinite(focusSamples) || focusSamples < COLD_SAMPLES) {
457
+ return { reactionMs, timeoutMs, cold: true, from: `fewer than ${COLD_SAMPLES} focus samples` };
458
+ }
459
+ const margin = Math.max(150, Math.round(focusP95 * 0.2));
460
+ const learnedTimeout = Math.min(HARD_CAP_MS, focusP95 + margin);
461
+ const learnedReaction = Math.min(learnedTimeout, focusP50 + margin);
462
+ return {
463
+ // max, not min. See above: only ever longer.
464
+ reactionMs: Math.max(reactionMs, learnedReaction),
465
+ timeoutMs: Math.max(timeoutMs, learnedTimeout),
466
+ cold: false,
467
+ from: `p95 ${focusP95}ms over ${focusSamples} focus samples`,
468
+ };
469
+ }
470
+
387
471
  /** What this edge's observed settle durations say, or that it has none. */
388
472
  export function timingOf(edge) {
389
473
  const samples = edge?.settles ?? [];
390
474
  const gaps = edge?.quietGaps ?? [];
475
+ const focuses = edge?.focuses ?? [];
391
476
  return {
392
477
  samples: samples.length,
393
478
  p50: metrics.percentile(samples, 50),
394
479
  p95: metrics.percentile(samples, 95),
395
480
  gapSamples: gaps.length,
396
481
  gapP95: metrics.percentile(gaps, 95),
482
+ // The same statistic measured after the transition rather than during it.
483
+ // Reported side by side so the size of the bias is visible rather than
484
+ // argued about — see `index.longestQuietGap`.
485
+ trueGapSamples: (edge?.trueGaps ?? []).length,
486
+ trueGapP95: metrics.percentile(edge?.trueGaps ?? [], 95),
487
+ focusSamples: focuses.length,
488
+ focusP50: metrics.percentile(focuses, 50),
489
+ focusP95: metrics.percentile(focuses, 95),
397
490
  };
398
491
  }
399
492
 
@@ -433,7 +526,7 @@ export function timingInto(udid, hash) {
433
526
  return best ? { ...timingOf(best), action: best.action, kind: best.kind ?? null } : null;
434
527
  }
435
528
 
436
- export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
529
+ export function record(udid, { from, action, to, kind, settleMs, quietGapMs, focusMs }) {
437
530
  const fromKey = typeof from === 'string' ? { hash: from } : from;
438
531
  const toHash = typeof to === 'string' ? to : to?.hash;
439
532
  if (!fromKey?.hash || !toHash) return null;
@@ -500,7 +593,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
500
593
  save(udid, target);
501
594
  existing.count += 1;
502
595
  existing.lastSeen = Date.now();
503
- noteSettle(existing, settleMs, quietGapMs);
596
+ noteSettle(existing, settleMs, quietGapMs, focusMs);
504
597
  save(udid, node);
505
598
  return node;
506
599
  }
@@ -512,7 +605,13 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
512
605
  existing.kind = kind ?? existing.kind;
513
606
  existing.count += 1;
514
607
  existing.lastSeen = Date.now();
515
- noteSettle(existing, settleMs);
608
+ // Every argument, and it is worth saying why this line once passed one.
609
+ // `quietGapMs` was dropped here — on the *main* path, the one nearly every
610
+ // recorded edge takes — so the pause statistic only ever accumulated on a
611
+ // brand-new edge and on the variant branch. The window that reads it looked
612
+ // permanently cold, which is a measurement quietly not being taken rather
613
+ // than a wrong number, and those are the ones nothing complains about.
614
+ noteSettle(existing, settleMs, quietGapMs, focusMs);
516
615
  } else {
517
616
  node.edges.push({
518
617
  action: signature,
@@ -525,6 +624,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
525
624
  lastSeen: Date.now(),
526
625
  settles: Number.isFinite(settleMs) && settleMs >= 0 ? [Math.round(settleMs)] : [],
527
626
  quietGaps: Number.isFinite(quietGapMs) && quietGapMs >= 0 ? [Math.round(quietGapMs)] : [],
627
+ focuses: Number.isFinite(focusMs) && focusMs >= 0 ? [Math.round(focusMs)] : [],
528
628
  });
529
629
  }
530
630
  save(udid, node);