simframe 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -3
- package/package.json +1 -1
- package/scripts/check-private.mjs +143 -0
- package/scripts/eval-perception.mjs +248 -0
- package/src/actions.js +150 -11
- package/src/analyze.js +70 -0
- package/src/cli.js +83 -6
- package/src/graph.js +104 -4
- package/src/index.js +218 -3
- package/src/input.js +49 -5
- package/src/matching.js +55 -2
- package/src/metrics.js +105 -8
- package/src/navigate.js +10 -7
- package/src/platform/android.js +1 -1
- package/src/screenmap.js +20 -1
- package/src/view.js +61 -0
package/src/actions.js
CHANGED
|
@@ -27,11 +27,22 @@ const MAX_PAUSE_MS = 5000;
|
|
|
27
27
|
* falls out at `reaction` instead. That bounds the cost of the honest case
|
|
28
28
|
* rather than the broken one.
|
|
29
29
|
*/
|
|
30
|
+
/**
|
|
31
|
+
* The stillness window stays fixed, and that is a decision rather than an
|
|
32
|
+
* oversight. Learning a stillness window is the half of Phase 11 that was
|
|
33
|
+
* reverted for cause: a wait that ends early never observes the pauses that
|
|
34
|
+
* come later, so the estimator ratchets itself down and the graph learns
|
|
35
|
+
* transitions that never happened. See docs/BENCHMARKS.md, Phase 11.
|
|
36
|
+
*/
|
|
30
37
|
const FOCUS_STABLE_MS = 250;
|
|
31
38
|
/**
|
|
32
|
-
*
|
|
33
|
-
*
|
|
34
|
-
*
|
|
39
|
+
* The cold defaults, unchanged, for a field this screen has not been measured
|
|
40
|
+
* focusing. `graph.focusPlan` takes over once it has been, and may only make
|
|
41
|
+
* the wait longer.
|
|
42
|
+
*
|
|
43
|
+
* 900 ms is long enough for a slow capture loop to produce a frame or two. The
|
|
44
|
+
* screenshot engine idles at 1.5 fps — 667 ms between frames — so anything
|
|
45
|
+
* under that is a verdict reached before there was anything to look at.
|
|
35
46
|
*/
|
|
36
47
|
const FOCUS_REACTION_MS = 900;
|
|
37
48
|
const FOCUS_TIMEOUT_MS = 3000;
|
|
@@ -140,7 +151,7 @@ export async function runScript(
|
|
|
140
151
|
// than no instrumentation.
|
|
141
152
|
const noteEscalation = (record) => {
|
|
142
153
|
try {
|
|
143
|
-
escalations.push(metrics.recordEscalation(udid, { flowId, ...record }));
|
|
154
|
+
escalations.push(metrics.recordEscalation(udid, { flowId, flowName, ...record }));
|
|
144
155
|
} catch {
|
|
145
156
|
/* instrumentation must not be able to fail a flow it is only watching */
|
|
146
157
|
}
|
|
@@ -171,10 +182,35 @@ export async function runScript(
|
|
|
171
182
|
// moves nothing and is not a failure, so an unbounded retry would rebuild the
|
|
172
183
|
// session and press again on every such step for no reason.
|
|
173
184
|
let inputRecovered = false;
|
|
185
|
+
/**
|
|
186
|
+
* The previous step's transition, still to be measured.
|
|
187
|
+
*
|
|
188
|
+
* Its pause profile cannot be read while the step is running — that is the
|
|
189
|
+
* biased measurement that corrupted the graph — so it is read one step later,
|
|
190
|
+
* off a frame history whose end nothing about the wait decided. See
|
|
191
|
+
* `api.longestQuietGap`.
|
|
192
|
+
*/
|
|
193
|
+
let pendingGap = null;
|
|
194
|
+
const measurePendingGap = async () => {
|
|
195
|
+
if (!pendingGap) return;
|
|
196
|
+
const { from, step: prevStep, actionAt } = pendingGap;
|
|
197
|
+
pendingGap = null;
|
|
198
|
+
try {
|
|
199
|
+
const history = (await api.getState(deviceQuery, { options })).state.history ?? [];
|
|
200
|
+
const trueGapMs = api.longestQuietGap(history, actionAt);
|
|
201
|
+
if (trueGapMs != null) graph.noteTrueGap(udid, from, prevStep, trueGapMs);
|
|
202
|
+
} catch {
|
|
203
|
+
/* a statistic nothing acts on must never be able to fail a flow */
|
|
204
|
+
}
|
|
205
|
+
};
|
|
174
206
|
|
|
175
207
|
for (const [i, raw] of steps.entries()) {
|
|
176
208
|
const step = normalizeStep(raw);
|
|
177
209
|
const stepStart = Date.now();
|
|
210
|
+
// Before anything else, and before this step disturbs the screen: the
|
|
211
|
+
// previous transition is definitely over by now, so its true pause profile
|
|
212
|
+
// is readable.
|
|
213
|
+
await measurePendingGap();
|
|
178
214
|
// The baseline for "did the screen react" must predate the action itself.
|
|
179
215
|
const beforeState = (await api.getState(deviceQuery, { options })).state;
|
|
180
216
|
const before = beforeState.hash;
|
|
@@ -190,11 +226,24 @@ export async function runScript(
|
|
|
190
226
|
// What this action did last time it was taken here, if ever.
|
|
191
227
|
const prediction = verify && beforeScreen?.hash ? graph.predict(udid, beforeScreen, step) : null;
|
|
192
228
|
try {
|
|
193
|
-
let detail = await runStep(deviceQuery, udid, step, { screen, options, frames });
|
|
194
229
|
// How long this transition has cost before, on this screen, for this
|
|
195
230
|
// action. A cold edge gets the old fixed default and says so; a measured
|
|
196
231
|
// one gets p95 plus a margin. Research §7.
|
|
232
|
+
//
|
|
233
|
+
// Read before the step, not after, because one of the waits it informs
|
|
234
|
+
// happens *inside* the step: a `type into` taps the field and waits for
|
|
235
|
+
// focus before it types, and that wait used to be three constants.
|
|
197
236
|
const learned = verify && beforeScreen?.hash ? graph.timingFor(udid, beforeScreen, step) : null;
|
|
237
|
+
const focus = {
|
|
238
|
+
plan: graph.focusPlan(learned, {
|
|
239
|
+
reactionMs: FOCUS_REACTION_MS,
|
|
240
|
+
timeoutMs: FOCUS_TIMEOUT_MS,
|
|
241
|
+
stillnessMs: FOCUS_STABLE_MS,
|
|
242
|
+
keyboardUp: Boolean(beforeScreen?.keyboard),
|
|
243
|
+
}),
|
|
244
|
+
observedMs: null,
|
|
245
|
+
};
|
|
246
|
+
let detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
|
|
198
247
|
// How long this screen must hold still before it counts as settled.
|
|
199
248
|
//
|
|
200
249
|
// 500 ms was a constant paid by every step of every flow, and it is the
|
|
@@ -257,6 +306,15 @@ export async function runScript(
|
|
|
257
306
|
budgetMs,
|
|
258
307
|
stillnessMs: stillness,
|
|
259
308
|
quietGapMs: w.quietGapMs,
|
|
309
|
+
// The baseline had already finished moving when the wait began, so it
|
|
310
|
+
// was re-taken from the live screen. Surfaced because it means the
|
|
311
|
+
// step before this one had not finished when this one started.
|
|
312
|
+
staleBaseline: Boolean(w.staleBaseline),
|
|
313
|
+
// Something moved in one region only — a switch, a radio dot, a
|
|
314
|
+
// segment highlight. Worth saying, because it is the difference
|
|
315
|
+
// between "the action did nothing" and "the action did something the
|
|
316
|
+
// whole-screen mean cannot see".
|
|
317
|
+
smallChange: Boolean(w.smallChange),
|
|
260
318
|
timing: learned
|
|
261
319
|
? {
|
|
262
320
|
p50: learned.p50,
|
|
@@ -343,7 +401,23 @@ export async function runScript(
|
|
|
343
401
|
// for. Requiring both meant a screen that settled slowly recorded
|
|
344
402
|
// nothing at all.
|
|
345
403
|
endScreen = afterScreen;
|
|
346
|
-
|
|
404
|
+
// An action with no observed effect teaches the graph nothing, and
|
|
405
|
+
// recording it teaches something false.
|
|
406
|
+
//
|
|
407
|
+
// This is the second half of the same bug. A settle that returned on a
|
|
408
|
+
// stale baseline reported `ok` for a tap that moved nothing, and the
|
|
409
|
+
// recorder asked only whether the *reading* was confirmed — so
|
|
410
|
+
// `root -> root` went in as a verified edge and started being
|
|
411
|
+
// predicted. Re-baselining stops the settle lying; this stops the
|
|
412
|
+
// graph learning from a step that has no evidence behind it either way.
|
|
413
|
+
//
|
|
414
|
+
// It does cost a real case for now: a control that genuinely returns to
|
|
415
|
+
// the same screen — a toggle — is invisible to the change detector at
|
|
416
|
+
// eight times below its threshold, so it reads as no-visible-change and
|
|
417
|
+
// its edge is no longer recorded. That is the right trade while the
|
|
418
|
+
// detector cannot see it, and it comes back on its own once it can.
|
|
419
|
+
const noEvidence = Boolean(settled?.noVisibleChange);
|
|
420
|
+
if (afterScreen.confirmed && afterScreen.hash && !noEvidence) {
|
|
347
421
|
// The observed cost of this transition, which is what makes the next
|
|
348
422
|
// one adaptive. Only from a settle that was actually satisfied: a
|
|
349
423
|
// timeout is not a measurement of how long the screen takes, it is a
|
|
@@ -356,13 +430,37 @@ export async function runScript(
|
|
|
356
430
|
// 400ms and then timed out is exactly the case a 150ms stillness
|
|
357
431
|
// window would have got wrong.
|
|
358
432
|
quietGapMs: settled?.sawChange ? settled.quietGapMs : undefined,
|
|
433
|
+
// The focus wait's own distribution, kept apart from the step's.
|
|
434
|
+
// Only set when a field was tapped and visibly took focus.
|
|
435
|
+
focusMs: focus.observedMs ?? undefined,
|
|
359
436
|
});
|
|
437
|
+
pendingGap = { from: beforeScreen, step, actionAt: stepStart };
|
|
438
|
+
carriedScreen = afterScreen;
|
|
439
|
+
} else if (afterScreen.confirmed && afterScreen.hash) {
|
|
440
|
+
// Where we are is still known; only what got us here is not worth
|
|
441
|
+
// remembering. Carrying it saves the next step a perception pass.
|
|
360
442
|
carriedScreen = afterScreen;
|
|
361
443
|
}
|
|
362
444
|
}
|
|
363
445
|
|
|
364
446
|
const wrongTurn = wrongTurnFrom(verification);
|
|
365
|
-
|
|
447
|
+
// `[no visible change]` after a launch is ambiguous between two very
|
|
448
|
+
// different things, and a real session read it the wrong way twice:
|
|
449
|
+
// "the app was already in front, so nothing needed to move" and "the app
|
|
450
|
+
// did not come forward". Measured on this Xcode, `simctl launch` *does*
|
|
451
|
+
// front an already-running app, so the first reading is the likely one —
|
|
452
|
+
// but likely is not the same as said, and the step is the only place that
|
|
453
|
+
// can say it.
|
|
454
|
+
const launchNote = step.action === 'launch' && settled?.noVisibleChange
|
|
455
|
+
? ' [the screen did not change, so this app was already in front — or it did not come forward]'
|
|
456
|
+
: '';
|
|
457
|
+
const note = launchNote
|
|
458
|
+
+ (settled?.smallChange ? ' [a small change, in one region only]' : '')
|
|
459
|
+
+ (settled?.noVisibleChange ? ' [no visible change]' : '')
|
|
460
|
+
+ (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
|
|
461
|
+
+ (settled?.blackFrames
|
|
462
|
+
? ` [${settled.blackFrames} black frame(s) waited through${settled.blackMs ? `, still black after ${settled.blackMs}ms` : ''}]`
|
|
463
|
+
: '');
|
|
366
464
|
results.push({
|
|
367
465
|
index: i,
|
|
368
466
|
action: step.action,
|
|
@@ -416,6 +514,10 @@ export async function runScript(
|
|
|
416
514
|
}
|
|
417
515
|
}
|
|
418
516
|
|
|
517
|
+
// The last step has no next step to measure it, and its transition is over by
|
|
518
|
+
// the time the loop exits.
|
|
519
|
+
await measurePendingGap();
|
|
520
|
+
|
|
419
521
|
const wallMs = Date.now() - startedAt;
|
|
420
522
|
try {
|
|
421
523
|
metrics.recordFlow(udid, metrics.flowRecordFrom({
|
|
@@ -464,18 +566,31 @@ export async function runScript(
|
|
|
464
566
|
*/
|
|
465
567
|
async function focusField(deviceQuery, udid, step, ctx) {
|
|
466
568
|
const found = await api.locate(deviceQuery, step.into, { index: step.index, refresh: step.refresh });
|
|
569
|
+
// What this field has cost to focus before, on this screen. Cold, or with no
|
|
570
|
+
// verification running, that is exactly the three constants above; measured,
|
|
571
|
+
// it can only be longer. `graph.focusPlan` carries the reason it is either.
|
|
572
|
+
const plan = ctx.focus?.plan ?? {
|
|
573
|
+
reactionMs: FOCUS_REACTION_MS, timeoutMs: FOCUS_TIMEOUT_MS, cold: true, from: 'no timing in hand',
|
|
574
|
+
};
|
|
575
|
+
const tappedAt = Date.now();
|
|
467
576
|
await input.tapPoint(udid, found.target.x, found.target.y);
|
|
468
577
|
const focused = await api.waitFor(deviceQuery, {
|
|
469
578
|
mode: 'settle',
|
|
470
579
|
stableMs: FOCUS_STABLE_MS,
|
|
471
|
-
reactionMs:
|
|
472
|
-
timeoutMs:
|
|
580
|
+
reactionMs: plan.reactionMs,
|
|
581
|
+
timeoutMs: plan.timeoutMs,
|
|
473
582
|
options: ctx.options,
|
|
474
583
|
});
|
|
584
|
+
// Only a wait that was satisfied is a measurement of how long focus takes. A
|
|
585
|
+
// reaction window that ran out measures how long we were prepared to watch a
|
|
586
|
+
// screen that did not move, and banking that would teach the edge the cost of
|
|
587
|
+
// its own impatience — the estimator mistake learned stillness made.
|
|
588
|
+
if (ctx.focus && focused.satisfied) ctx.focus.observedMs = Date.now() - tappedAt;
|
|
475
589
|
return {
|
|
476
590
|
found,
|
|
477
591
|
where: `"${found.target.label}" at ${found.target.x},${found.target.y}`,
|
|
478
592
|
quiet: focused.satisfied ? '' : ' [the field did not visibly take focus]',
|
|
593
|
+
waited: focused.satisfied && !plan.cold ? ` [focus in ${focused.waitedMs}ms, ${plan.from}]` : '',
|
|
479
594
|
};
|
|
480
595
|
}
|
|
481
596
|
|
|
@@ -507,7 +622,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
507
622
|
if (step.into) {
|
|
508
623
|
const field = await focusField(deviceQuery, udid, step, ctx);
|
|
509
624
|
await input.typeText(udid, step.text ?? step.value);
|
|
510
|
-
return `typed into ${field.where}${field.quiet}`;
|
|
625
|
+
return `typed into ${field.where}${field.quiet}${field.waited}`;
|
|
511
626
|
}
|
|
512
627
|
await input.typeText(udid, step.text ?? step.value);
|
|
513
628
|
return 'typed text';
|
|
@@ -520,7 +635,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
520
635
|
if (step.into) {
|
|
521
636
|
const field = await focusField(deviceQuery, udid, step, ctx);
|
|
522
637
|
await input.pasteText(udid, step.text ?? step.value);
|
|
523
|
-
return `pasted into ${field.where}${field.quiet}`;
|
|
638
|
+
return `pasted into ${field.where}${field.quiet}${field.waited}`;
|
|
524
639
|
}
|
|
525
640
|
await input.pasteText(udid, step.text ?? step.value);
|
|
526
641
|
return 'pasted into the focused field';
|
|
@@ -600,6 +715,14 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
600
715
|
return `"${node.label ?? target}" appeared`;
|
|
601
716
|
} catch (err) {
|
|
602
717
|
lastError = err.message;
|
|
718
|
+
// Same rule as `waitFor`, and `matchElement` says it in its own
|
|
719
|
+
// words: a query that matched several elements has found them all
|
|
720
|
+
// already.
|
|
721
|
+
if (/matched \d+ elements/.test(err.message)) {
|
|
722
|
+
throw new Error(
|
|
723
|
+
`${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
|
|
724
|
+
);
|
|
725
|
+
}
|
|
603
726
|
}
|
|
604
727
|
await sleep(250);
|
|
605
728
|
}
|
|
@@ -652,6 +775,22 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
652
775
|
return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`;
|
|
653
776
|
} catch (err) {
|
|
654
777
|
lastError = err.message;
|
|
778
|
+
// Waiting cannot make a thing unique.
|
|
779
|
+
//
|
|
780
|
+
// Reported from a real session: a wait on an ambiguous string spent
|
|
781
|
+
// the full 30 s and then listed four matches, all four of which were
|
|
782
|
+
// on the very first frame. The disambiguation is good and it arrived
|
|
783
|
+
// twenty-nine seconds after everything it needed. `ambiguous` means
|
|
784
|
+
// the target is *present*, several times over — which is precisely
|
|
785
|
+
// the case where more time changes nothing.
|
|
786
|
+
//
|
|
787
|
+
// Distinguished by the tag at the throw site rather than by reading
|
|
788
|
+
// the message, because "not on this screen" tags the same reason.
|
|
789
|
+
if (metrics.escalationOf(err)?.ambiguous) {
|
|
790
|
+
throw new Error(
|
|
791
|
+
`${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
|
|
792
|
+
);
|
|
793
|
+
}
|
|
655
794
|
}
|
|
656
795
|
if (Date.now() >= limit) break;
|
|
657
796
|
await sleep(POLL_MS);
|
package/src/analyze.js
CHANGED
|
@@ -35,6 +35,42 @@ export function signatureDiff(a, b) {
|
|
|
35
35
|
return sum / a.length / 255;
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
+
/**
|
|
39
|
+
* A change small enough that the whole-screen mean cannot see it.
|
|
40
|
+
*
|
|
41
|
+
* `changed` in both daemons is `signatureDiff > 0.004`, a mean over a 4x8 grid
|
|
42
|
+
* of gray means. Measured on this device, an iOS switch flipping:
|
|
43
|
+
*
|
|
44
|
+
* mean diff 0.001348 — a third of the threshold, so: not a change
|
|
45
|
+
* max cell delta 0.043137 — one cell of thirty-two, in row 1
|
|
46
|
+
*
|
|
47
|
+
* So the entire class of small binary controls — switches, radio dots,
|
|
48
|
+
* checkboxes, segment highlights — changes nothing as far as the daemon is
|
|
49
|
+
* concerned, and a step that flips one reports `no-visible-change`, which is a
|
|
50
|
+
* verdict that escalates.
|
|
51
|
+
*
|
|
52
|
+
* The threshold sits in a measured gap rather than being chosen. Eighty seconds
|
|
53
|
+
* of a *static* screen gave a largest per-cell delta of 0.003922, in row 0,
|
|
54
|
+
* which is the status-bar clock ticking over — the only thing moving. So the
|
|
55
|
+
* separation is 0.0039 against 0.0431, eleven times, and 0.012 is three times
|
|
56
|
+
* the noise and three and a half times under the signal. No row is excluded:
|
|
57
|
+
* the clock does not reach the threshold, which is a better reason to ignore it
|
|
58
|
+
* than a structural exclusion that would also blind the nav bar.
|
|
59
|
+
*
|
|
60
|
+
* What this deliberately does **not** do is feed stillness. `stableForMs` stays
|
|
61
|
+
* on the mean, because a blinking text caret is a small localised change and a
|
|
62
|
+
* screen with a cursor in it would otherwise never settle. The two signals are
|
|
63
|
+
* independent by design: this one answers "did the action do anything", and the
|
|
64
|
+
* mean answers "has the screen finished moving".
|
|
65
|
+
*/
|
|
66
|
+
export const CELL_CHANGE = 0.012;
|
|
67
|
+
|
|
68
|
+
/** The largest single-region change between two signatures, 0-1. */
|
|
69
|
+
export function maxCellDelta(a, b) {
|
|
70
|
+
const deltas = regionDeltas(a, b);
|
|
71
|
+
return deltas.length ? Math.max(...deltas) : 0;
|
|
72
|
+
}
|
|
73
|
+
|
|
38
74
|
/** Per-region change fractions, so callers can tell a toast from a screen push. */
|
|
39
75
|
export function regionDeltas(a, b) {
|
|
40
76
|
if (!a || !b || a.length !== b.length) return a ? a.map(() => 1) : [];
|
|
@@ -69,6 +105,40 @@ export function signatureToHex(sig) {
|
|
|
69
105
|
return sig.map((v) => v.toString(16).padStart(2, '0')).join('');
|
|
70
106
|
}
|
|
71
107
|
|
|
108
|
+
/**
|
|
109
|
+
* Is this frame black — not dark, black?
|
|
110
|
+
*
|
|
111
|
+
* The capture wedge (docs/BENCHMARKS.md) leaves `simctl io screenshot`
|
|
112
|
+
* succeeding and returning 0 non-black pixels of 3,162,132, and simframe's own
|
|
113
|
+
* frames go the same way: the display pipeline stops rendering while
|
|
114
|
+
* everything about the capture path keeps reporting success. It self-recovers
|
|
115
|
+
* most times and a device restart cures the rest, so the useful thing is to
|
|
116
|
+
* notice early — before a settle spends its whole budget deciding a black
|
|
117
|
+
* screen is a calm one.
|
|
118
|
+
*
|
|
119
|
+
* The signature is already computed for every frame, so this costs 32 integer
|
|
120
|
+
* comparisons and no decode. A real screen does not come close: measured on
|
|
121
|
+
* this device's Settings root, the 32 bytes ran 191–245.
|
|
122
|
+
*
|
|
123
|
+
* The threshold is a level, not a fraction, and it is on the *maximum*: one
|
|
124
|
+
* cell with anything in it is enough to say the display is rendering. That
|
|
125
|
+
* matters because a dark-mode screen, a video, or a splash on black are all
|
|
126
|
+
* legitimately near-zero in most cells and this must not call them faults.
|
|
127
|
+
*
|
|
128
|
+
* And it says "the frames are black", never "the simulator is wedged". A
|
|
129
|
+
* screen can be black because the app drew black. What makes it a wedge is
|
|
130
|
+
* that it stays black while input is being delivered, and only the caller
|
|
131
|
+
* knows that.
|
|
132
|
+
*/
|
|
133
|
+
export const BLACK_LEVEL = 8;
|
|
134
|
+
|
|
135
|
+
export function isBlackFrame(sig, { level = BLACK_LEVEL } = {}) {
|
|
136
|
+
const bytes = typeof sig === 'string' ? hexToSignature(sig) : sig;
|
|
137
|
+
if (!bytes?.length) return false;
|
|
138
|
+
for (const b of bytes) if (b > level) return false;
|
|
139
|
+
return true;
|
|
140
|
+
}
|
|
141
|
+
|
|
72
142
|
export function hexToSignature(hex) {
|
|
73
143
|
const out = [];
|
|
74
144
|
for (let i = 0; i < hex.length; i += 2) out.push(parseInt(hex.slice(i, i + 2), 16));
|
package/src/cli.js
CHANGED
|
@@ -5,6 +5,7 @@ import path from 'node:path';
|
|
|
5
5
|
import { runDaemon, DEFAULTS } from './daemon.js';
|
|
6
6
|
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
|
|
7
7
|
import * as actions from './actions.js';
|
|
8
|
+
import * as analyze from './analyze.js';
|
|
8
9
|
import * as api from './index.js';
|
|
9
10
|
import * as input from './input.js';
|
|
10
11
|
import * as baseline from './baseline.js';
|
|
@@ -20,6 +21,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
20
21
|
simframe start [device] start the capture loop in the background
|
|
21
22
|
simframe stop [device|--all] stop the capture loop
|
|
22
23
|
simframe status [device] show daemon and newest-frame status
|
|
24
|
+
simframe input reset rebuild the HID session (see doctor)
|
|
23
25
|
simframe frame [device] write the newest frame to a file
|
|
24
26
|
simframe state [device] print frame metadata and the change map
|
|
25
27
|
simframe mark [device] print the current frame hash, to use as --since
|
|
@@ -310,6 +312,29 @@ async function main() {
|
|
|
310
312
|
return;
|
|
311
313
|
}
|
|
312
314
|
|
|
315
|
+
// Rebuild the daemon's HID session, and nothing else.
|
|
316
|
+
//
|
|
317
|
+
// The narrow remedy for the narrow fault. Restarting the daemon also cures
|
|
318
|
+
// a stale session and throws away the frame ring and every warm cache to do
|
|
319
|
+
// it, which is the difference between a fix and a power cycle.
|
|
320
|
+
case 'input': {
|
|
321
|
+
const what = positional[0];
|
|
322
|
+
if (what !== 'reset') throw new Error('usage: simframe input reset [--device <udid>]');
|
|
323
|
+
const dev = await resolveDevice(device);
|
|
324
|
+
const before = await input.sessionHealth(dev.udid);
|
|
325
|
+
const reset = await input.resetSession(dev.udid);
|
|
326
|
+
if (!reset) {
|
|
327
|
+
// Said plainly rather than as a success: there is no daemon holding a
|
|
328
|
+
// session to rebuild, so nothing was wrong and nothing was done.
|
|
329
|
+
console.log(`no simframed session to rebuild for ${dev.name} — the daemon is not running, or this device is not driven by it`);
|
|
330
|
+
process.exitCode = 1;
|
|
331
|
+
return;
|
|
332
|
+
}
|
|
333
|
+
console.log(`rebuilt the HID session for ${dev.name}`
|
|
334
|
+
+ (before.stale ? `\n it was stale: ${before.reason}` : '\n it did not report stale; rebuilt anyway, as asked'));
|
|
335
|
+
return;
|
|
336
|
+
}
|
|
337
|
+
|
|
313
338
|
case 'status': {
|
|
314
339
|
const udids = device
|
|
315
340
|
? [(await resolveDevice(device)).udid]
|
|
@@ -951,23 +976,52 @@ async function main() {
|
|
|
951
976
|
case 'escalations': {
|
|
952
977
|
const dev = await resolveDevice(flags.device);
|
|
953
978
|
const records = metrics.readEscalations(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
|
|
954
|
-
const b = metrics.breakdown(records
|
|
979
|
+
const b = metrics.breakdown(records, {
|
|
980
|
+
session: flags.session === true ? metrics.sessionId() : (flags.session ? String(flags.session) : null),
|
|
981
|
+
flow: flags.flow ? String(flags.flow) : null,
|
|
982
|
+
});
|
|
955
983
|
if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify(b, null, 2)}\n`);
|
|
956
984
|
emit(flags, b, [
|
|
957
985
|
`${b.total} escalation${b.total === 1 ? '' : 's'} on ${dev.name}`,
|
|
958
986
|
...metrics.REASONS
|
|
959
987
|
.filter((r) => b.by_reason[r])
|
|
960
988
|
.sort((a, c) => b.by_reason[c] - b.by_reason[a])
|
|
961
|
-
.map((r) => ` ${r.padEnd(20)} ${String(b.by_reason[r]).padStart(4)}
|
|
989
|
+
.map((r) => ` ${r.padEnd(20)} ${String(b.by_reason[r]).padStart(4)} `
|
|
990
|
+
+ (metrics.BUILT_FACULTIES.has(metrics.FACULTY[r])
|
|
991
|
+
? `not removed by: ${metrics.FACULTY[r]} [built]`
|
|
992
|
+
: `would be removed by: ${metrics.FACULTY[r]}`)),
|
|
962
993
|
b.total ? '' : null,
|
|
963
994
|
b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
|
|
964
995
|
// Said out loud rather than left for someone to discover: the rate is
|
|
965
996
|
// 1.0 while no faculty exists, so the breakdown above is the number
|
|
966
997
|
// that decides the next phase.
|
|
967
998
|
b.total && b.avoidable_escalation_rate === 1
|
|
968
|
-
?
|
|
999
|
+
? (metrics.REASONS.some((r) => b.by_reason[r] && metrics.BUILT_FACULTIES.has(metrics.FACULTY[r]))
|
|
1000
|
+
? ' the rate is 1.0 because nothing resolves locally yet. A reason marked [built] is not a queue waiting on a phase — it is evidence the phase that shipped is not sufficient.'
|
|
1001
|
+
: ' every reason maps to a faculty that is not built yet, so this rate is 1.0 by construction. The per-reason counts are the steering wheel.')
|
|
969
1002
|
: null,
|
|
970
1003
|
b.total ? `model turns spent on escalations: ${b.model_turns_spent}` : null,
|
|
1004
|
+
// The log is per-device and shared. Said before the counts are used,
|
|
1005
|
+
// not after: two agents on one booted simulator write one interleaved
|
|
1006
|
+
// file, and CLAUDE.md makes these counts the thing that picks the next
|
|
1007
|
+
// faculty. A pooled breakdown errs toward whichever session made more
|
|
1008
|
+
// mistakes, which is a different question.
|
|
1009
|
+
b.pooled
|
|
1010
|
+
? 'WARNING these counts may pool more than one agent\'s work: '
|
|
1011
|
+
+ [
|
|
1012
|
+
b.session_count > 1 ? `${b.session_count} sessions` : null,
|
|
1013
|
+
b.unattributed ? `${b.unattributed} record(s) written before sessions were logged` : null,
|
|
1014
|
+
].filter(Boolean).join(', ')
|
|
1015
|
+
+ '. Narrow with --session (this process), --session=<id>, or --flow=<name>.'
|
|
1016
|
+
: null,
|
|
1017
|
+
b.session_count > 1 ? 'sessions:' : null,
|
|
1018
|
+
...(b.session_count > 1
|
|
1019
|
+
? b.sessions.map((x) => ` ${x.session_id.padEnd(22)} ${String(x.count).padStart(4)} ${x.client}`)
|
|
1020
|
+
: []),
|
|
1021
|
+
Object.keys(b.by_flow).length > 1 ? 'flows:' : null,
|
|
1022
|
+
...(Object.keys(b.by_flow).length > 1
|
|
1023
|
+
? Object.entries(b.by_flow).slice(0, 10).map(([n, c]) => ` ${n.padEnd(28)} ${String(c).padStart(4)}`)
|
|
1024
|
+
: []),
|
|
971
1025
|
b.top_screens.length ? 'top screens:' : null,
|
|
972
1026
|
...b.top_screens.map((s) => ` ${s.fingerprint.slice(0, 16).padEnd(18)} ${s.count}`),
|
|
973
1027
|
metrics.writeError() ? `WARNING a log write failed: ${metrics.writeError()}` : null,
|
|
@@ -1228,7 +1282,18 @@ async function doctor({ json = false, strict = false, device } = {}) {
|
|
|
1228
1282
|
// dispatched successfully and moved nothing, five runs in a row.
|
|
1229
1283
|
const session = await input.sessionHealth(d.udid);
|
|
1230
1284
|
if (session.stale) {
|
|
1231
|
-
|
|
1285
|
+
// The remedy used to read "the next action rebuilds it automatically;
|
|
1286
|
+
// simframe stop && simframe start does it now", and both halves were
|
|
1287
|
+
// wrong. The first was false wherever it mattered, because the rebuild
|
|
1288
|
+
// check was gated once per process and the MCP server is one process
|
|
1289
|
+
// for a whole session. The second names a command that fails twice:
|
|
1290
|
+
// `stop` needs `--device` when two simulators are booted, and then
|
|
1291
|
+
// refuses because a client holds the daemon, so the sequence that
|
|
1292
|
+
// actually works is `stop --device <udid> --force && start --device
|
|
1293
|
+
// <udid>` — a daemon restart, to fix a session, when rebuilding the
|
|
1294
|
+
// session is a thing the daemon can already do on request. It just had
|
|
1295
|
+
// no way in from outside. It does now.
|
|
1296
|
+
add(`input session (${d.name})`, 'warn', `stale — ${session.reason}. The next action rebuilds it; simframe input reset --device ${d.udid} does it now`,
|
|
1232
1297
|
{ key: 'input.session', value: 'stale' });
|
|
1233
1298
|
} else if (caps.input.supported) {
|
|
1234
1299
|
add(`input session (${d.name})`, 'ok', session.reason ?? 'current with this device session',
|
|
@@ -1252,8 +1317,20 @@ async function doctor({ json = false, strict = false, device } = {}) {
|
|
|
1252
1317
|
if (probed.length) {
|
|
1253
1318
|
const t0 = Date.now();
|
|
1254
1319
|
const res = await api.getFrame(probed[0].udid);
|
|
1255
|
-
|
|
1256
|
-
|
|
1320
|
+
// The wedge's whole signature is that everything here reports success.
|
|
1321
|
+
// A black frame is 32 integer comparisons on a signature already
|
|
1322
|
+
// computed, and it is what separates "captured a frame" from "captured
|
|
1323
|
+
// a frame of a display that has stopped rendering".
|
|
1324
|
+
const dark = analyze.isBlackFrame(
|
|
1325
|
+
(res.state.history ?? []).find((h) => h.seq === res.state.seq)?.sig ?? null,
|
|
1326
|
+
);
|
|
1327
|
+
add('capture', dark ? 'warn' : 'ok',
|
|
1328
|
+
`frame #${res.state.seq} ${res.width}x${res.height} in ${Date.now() - t0}ms (age ${res.ageMs}ms)`
|
|
1329
|
+
+ (dark
|
|
1330
|
+
? ' — and every pixel of it is black. If the device is not showing a black screen on purpose,'
|
|
1331
|
+
+ ' this is the display pipeline having stopped rendering; it usually recovers on its own,'
|
|
1332
|
+
+ ` and ${'xcrun simctl shutdown'} / boot is the cure that always works.`
|
|
1333
|
+
: ''),
|
|
1257
1334
|
{ key: 'capture.frames', value: res.state.seq });
|
|
1258
1335
|
// A wedged device produces the same nothing as a quiet one, so doctor has
|
|
1259
1336
|
// to ask the capture loop rather than look at the frames. `fail`, not
|
package/src/graph.js
CHANGED
|
@@ -348,7 +348,7 @@ export function findScreen(udid, query) {
|
|
|
348
348
|
* 2.4 s on a screen that fetches — and because the graph is already persisted,
|
|
349
349
|
* versioned and pruned.
|
|
350
350
|
*/
|
|
351
|
-
function noteSettle(edge, settleMs, quietGapMs) {
|
|
351
|
+
function noteSettle(edge, settleMs, quietGapMs, focusMs) {
|
|
352
352
|
if (Number.isFinite(settleMs) && settleMs >= 0) {
|
|
353
353
|
edge.settles = [...(edge.settles ?? []), Math.round(settleMs)].slice(-TIMING_WINDOW);
|
|
354
354
|
}
|
|
@@ -357,6 +357,39 @@ function noteSettle(edge, settleMs, quietGapMs) {
|
|
|
357
357
|
if (Number.isFinite(quietGapMs) && quietGapMs >= 0) {
|
|
358
358
|
edge.quietGaps = [...(edge.quietGaps ?? []), Math.round(quietGapMs)].slice(-TIMING_WINDOW);
|
|
359
359
|
}
|
|
360
|
+
// How long the *field* took to take focus, which is a different duration from
|
|
361
|
+
// how long the step took: it is measured between the tap and the keyboard,
|
|
362
|
+
// inside a step whose settle is measured after the typing. One edge, two
|
|
363
|
+
// waits, so two distributions.
|
|
364
|
+
if (Number.isFinite(focusMs) && focusMs >= 0) {
|
|
365
|
+
edge.focuses = [...(edge.focuses ?? []), Math.round(focusMs)].slice(-TIMING_WINDOW);
|
|
366
|
+
}
|
|
367
|
+
}
|
|
368
|
+
|
|
369
|
+
/**
|
|
370
|
+
* Record the unbiased pause statistic for an edge already written.
|
|
371
|
+
*
|
|
372
|
+
* Separate from `record` because it arrives later on purpose. The biased
|
|
373
|
+
* `quietGaps` are gathered from inside the wait and are therefore bounded by
|
|
374
|
+
* when the wait chose to stop; `trueGaps` are read off the frame history once
|
|
375
|
+
* the transition is definitely over, which is one step later. So the edge has
|
|
376
|
+
* to be found again rather than passed along.
|
|
377
|
+
*
|
|
378
|
+
* Nothing reads `trueGaps` yet, and that is deliberate. Phase 11 built the
|
|
379
|
+
* learned stillness window on the biased statistic, it corrupted the graph
|
|
380
|
+
* inside an afternoon, and the lesson taken was not "use a better estimator" —
|
|
381
|
+
* it was that a number gets to *act* only after it has been watched for a while
|
|
382
|
+
* doing nothing. This is the watching.
|
|
383
|
+
*/
|
|
384
|
+
export function noteTrueGap(udid, screen, step, trueGapMs) {
|
|
385
|
+
if (!Number.isFinite(trueGapMs) || trueGapMs < 0) return null;
|
|
386
|
+
const node = screen?.hash ? nearestScreen(udid, screen)?.node : null;
|
|
387
|
+
if (!node) return null;
|
|
388
|
+
const edge = node.edges?.find((e) => e.action === actionSignature(step));
|
|
389
|
+
if (!edge) return null;
|
|
390
|
+
edge.trueGaps = [...(edge.trueGaps ?? []), Math.round(trueGapMs)].slice(-TIMING_WINDOW);
|
|
391
|
+
save(udid, node);
|
|
392
|
+
return edge;
|
|
360
393
|
}
|
|
361
394
|
|
|
362
395
|
/**
|
|
@@ -384,16 +417,76 @@ export function stillnessFor({ gapSamples, gapP95 } = {}, fallbackMs) {
|
|
|
384
417
|
};
|
|
385
418
|
}
|
|
386
419
|
|
|
420
|
+
/**
|
|
421
|
+
* How long to wait for a tapped field to take focus.
|
|
422
|
+
*
|
|
423
|
+
* The three numbers this replaces were the last genuinely fixed waits on the
|
|
424
|
+
* action path: 250 ms of stillness, a 900 ms reaction window, a 3 s timeout.
|
|
425
|
+
* What makes them different from the step budget is the shape of the failure.
|
|
426
|
+
* A step budget that is too short reports `no-visible-change` and the flow can
|
|
427
|
+
* see it. A focus wait that is too short types into a field that does not have
|
|
428
|
+
* focus yet, `typeText` succeeds because input has no feedback channel, and the
|
|
429
|
+
* step reports that it typed — the worst shape a failure can take, and the bug
|
|
430
|
+
* this helper was written to fix in the first place.
|
|
431
|
+
*
|
|
432
|
+
* So this one is asymmetric on purpose: **a learned window may only lengthen
|
|
433
|
+
* the wait**, never shorten it. p95 of what this field has actually cost, when
|
|
434
|
+
* that is longer than 3 s, is a field that was being typed into too early and
|
|
435
|
+
* now is not. Where it is shorter, the measurement is discarded rather than
|
|
436
|
+
* banked as a saving — 5% of a distribution is one silent wrong type in twenty
|
|
437
|
+
* runs, and there is no amount of median wall time worth that.
|
|
438
|
+
*
|
|
439
|
+
* The one shortening is not a learned number at all, it is positive evidence:
|
|
440
|
+
* if the keyboard was already up before the tap, this tap moves a caret. There
|
|
441
|
+
* is no keyboard animation to wait for, so the reaction window collapses to the
|
|
442
|
+
* stillness window instead of paying 900 ms to watch a screen that was never
|
|
443
|
+
* going to move. `beforeScreen` already carries `keyboard`, so the evidence is
|
|
444
|
+
* free — it is the same perception pass the step was going to run anyway.
|
|
445
|
+
*/
|
|
446
|
+
export function focusPlan(stats, { reactionMs, timeoutMs, stillnessMs, keyboardUp } = {}) {
|
|
447
|
+
if (keyboardUp) {
|
|
448
|
+
return {
|
|
449
|
+
reactionMs: stillnessMs ?? reactionMs,
|
|
450
|
+
timeoutMs,
|
|
451
|
+
cold: false,
|
|
452
|
+
from: 'the keyboard was already up, so this tap moves a caret',
|
|
453
|
+
};
|
|
454
|
+
}
|
|
455
|
+
const { focusP50, focusP95, focusSamples } = stats ?? {};
|
|
456
|
+
if (!Number.isFinite(focusP95) || !Number.isFinite(focusSamples) || focusSamples < COLD_SAMPLES) {
|
|
457
|
+
return { reactionMs, timeoutMs, cold: true, from: `fewer than ${COLD_SAMPLES} focus samples` };
|
|
458
|
+
}
|
|
459
|
+
const margin = Math.max(150, Math.round(focusP95 * 0.2));
|
|
460
|
+
const learnedTimeout = Math.min(HARD_CAP_MS, focusP95 + margin);
|
|
461
|
+
const learnedReaction = Math.min(learnedTimeout, focusP50 + margin);
|
|
462
|
+
return {
|
|
463
|
+
// max, not min. See above: only ever longer.
|
|
464
|
+
reactionMs: Math.max(reactionMs, learnedReaction),
|
|
465
|
+
timeoutMs: Math.max(timeoutMs, learnedTimeout),
|
|
466
|
+
cold: false,
|
|
467
|
+
from: `p95 ${focusP95}ms over ${focusSamples} focus samples`,
|
|
468
|
+
};
|
|
469
|
+
}
|
|
470
|
+
|
|
387
471
|
/** What this edge's observed settle durations say, or that it has none. */
|
|
388
472
|
export function timingOf(edge) {
|
|
389
473
|
const samples = edge?.settles ?? [];
|
|
390
474
|
const gaps = edge?.quietGaps ?? [];
|
|
475
|
+
const focuses = edge?.focuses ?? [];
|
|
391
476
|
return {
|
|
392
477
|
samples: samples.length,
|
|
393
478
|
p50: metrics.percentile(samples, 50),
|
|
394
479
|
p95: metrics.percentile(samples, 95),
|
|
395
480
|
gapSamples: gaps.length,
|
|
396
481
|
gapP95: metrics.percentile(gaps, 95),
|
|
482
|
+
// The same statistic measured after the transition rather than during it.
|
|
483
|
+
// Reported side by side so the size of the bias is visible rather than
|
|
484
|
+
// argued about — see `index.longestQuietGap`.
|
|
485
|
+
trueGapSamples: (edge?.trueGaps ?? []).length,
|
|
486
|
+
trueGapP95: metrics.percentile(edge?.trueGaps ?? [], 95),
|
|
487
|
+
focusSamples: focuses.length,
|
|
488
|
+
focusP50: metrics.percentile(focuses, 50),
|
|
489
|
+
focusP95: metrics.percentile(focuses, 95),
|
|
397
490
|
};
|
|
398
491
|
}
|
|
399
492
|
|
|
@@ -433,7 +526,7 @@ export function timingInto(udid, hash) {
|
|
|
433
526
|
return best ? { ...timingOf(best), action: best.action, kind: best.kind ?? null } : null;
|
|
434
527
|
}
|
|
435
528
|
|
|
436
|
-
export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
529
|
+
export function record(udid, { from, action, to, kind, settleMs, quietGapMs, focusMs }) {
|
|
437
530
|
const fromKey = typeof from === 'string' ? { hash: from } : from;
|
|
438
531
|
const toHash = typeof to === 'string' ? to : to?.hash;
|
|
439
532
|
if (!fromKey?.hash || !toHash) return null;
|
|
@@ -500,7 +593,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
|
500
593
|
save(udid, target);
|
|
501
594
|
existing.count += 1;
|
|
502
595
|
existing.lastSeen = Date.now();
|
|
503
|
-
noteSettle(existing, settleMs, quietGapMs);
|
|
596
|
+
noteSettle(existing, settleMs, quietGapMs, focusMs);
|
|
504
597
|
save(udid, node);
|
|
505
598
|
return node;
|
|
506
599
|
}
|
|
@@ -512,7 +605,13 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
|
512
605
|
existing.kind = kind ?? existing.kind;
|
|
513
606
|
existing.count += 1;
|
|
514
607
|
existing.lastSeen = Date.now();
|
|
515
|
-
|
|
608
|
+
// Every argument, and it is worth saying why this line once passed one.
|
|
609
|
+
// `quietGapMs` was dropped here — on the *main* path, the one nearly every
|
|
610
|
+
// recorded edge takes — so the pause statistic only ever accumulated on a
|
|
611
|
+
// brand-new edge and on the variant branch. The window that reads it looked
|
|
612
|
+
// permanently cold, which is a measurement quietly not being taken rather
|
|
613
|
+
// than a wrong number, and those are the ones nothing complains about.
|
|
614
|
+
noteSettle(existing, settleMs, quietGapMs, focusMs);
|
|
516
615
|
} else {
|
|
517
616
|
node.edges.push({
|
|
518
617
|
action: signature,
|
|
@@ -525,6 +624,7 @@ export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
|
525
624
|
lastSeen: Date.now(),
|
|
526
625
|
settles: Number.isFinite(settleMs) && settleMs >= 0 ? [Math.round(settleMs)] : [],
|
|
527
626
|
quietGaps: Number.isFinite(quietGapMs) && quietGapMs >= 0 ? [Math.round(quietGapMs)] : [],
|
|
627
|
+
focuses: Number.isFinite(focusMs) && focusMs >= 0 ? [Math.round(focusMs)] : [],
|
|
528
628
|
});
|
|
529
629
|
}
|
|
530
630
|
save(udid, node);
|