simframe 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -3
- package/flows/hpi-suite.json +68 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +49 -1
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +7 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +11 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +48 -0
- package/native/simframed/Sources/simframed/main.swift +128 -73
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +40 -0
- package/package.json +2 -1
- package/scripts/bench-hpi.mjs +254 -0
- package/scripts/check-package.mjs +7 -0
- package/scripts/check-private.mjs +143 -0
- package/scripts/eval-perception.mjs +248 -0
- package/src/actions.js +333 -14
- package/src/analyze.js +70 -0
- package/src/baseline.js +333 -0
- package/src/cli.js +410 -4
- package/src/daemon.js +9 -0
- package/src/fingerprint.js +7 -1
- package/src/graph.js +262 -1
- package/src/index.js +335 -22
- package/src/input.js +155 -1
- package/src/intent.js +11 -2
- package/src/matching.js +136 -4
- package/src/mcp.js +14 -1
- package/src/metrics.js +596 -0
- package/src/navigate.js +47 -7
- package/src/platform/android.js +16 -1
- package/src/platform/index.js +3 -0
- package/src/platform/ios.js +40 -0
- package/src/screenmap.js +55 -14
- package/src/view.js +65 -3
package/src/actions.js
CHANGED
|
@@ -6,6 +6,8 @@ import * as api from './index.js';
|
|
|
6
6
|
import * as graph from './graph.js';
|
|
7
7
|
import * as input from './input.js';
|
|
8
8
|
import * as intent from './intent.js';
|
|
9
|
+
import * as metrics from './metrics.js';
|
|
10
|
+
import * as screenmap from './screenmap.js';
|
|
9
11
|
import { launchApp, openUrl, setPermission, terminateApp } from './platform/index.js';
|
|
10
12
|
|
|
11
13
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
@@ -25,11 +27,22 @@ const MAX_PAUSE_MS = 5000;
|
|
|
25
27
|
* falls out at `reaction` instead. That bounds the cost of the honest case
|
|
26
28
|
* rather than the broken one.
|
|
27
29
|
*/
|
|
30
|
+
/**
|
|
31
|
+
* The stillness window stays fixed, and that is a decision rather than an
|
|
32
|
+
* oversight. Learning a stillness window is the half of Phase 11 that was
|
|
33
|
+
* reverted for cause: a wait that ends early never observes the pauses that
|
|
34
|
+
* come later, so the estimator ratchets itself down and the graph learns
|
|
35
|
+
* transitions that never happened. See docs/BENCHMARKS.md, Phase 11.
|
|
36
|
+
*/
|
|
28
37
|
const FOCUS_STABLE_MS = 250;
|
|
29
38
|
/**
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
39
|
+
* The cold defaults, unchanged, for a field this screen has not been measured
|
|
40
|
+
* focusing. `graph.focusPlan` takes over once it has been, and may only make
|
|
41
|
+
* the wait longer.
|
|
42
|
+
*
|
|
43
|
+
* 900 ms is long enough for a slow capture loop to produce a frame or two. The
|
|
44
|
+
* screenshot engine idles at 1.5 fps — 667 ms between frames — so anything
|
|
45
|
+
* under that is a verdict reached before there was anything to look at.
|
|
33
46
|
*/
|
|
34
47
|
const FOCUS_REACTION_MS = 900;
|
|
35
48
|
const FOCUS_TIMEOUT_MS = 3000;
|
|
@@ -112,6 +125,12 @@ export async function runScript(
|
|
|
112
125
|
// Rebuild the HID session and retry once when a hardware button provably
|
|
113
126
|
// did nothing. Off only for a caller deliberately testing that path.
|
|
114
127
|
recoverInput = true,
|
|
128
|
+
// What this run is called and how few steps it could take, for the flow
|
|
129
|
+
// record. A bare `sim_do` has neither and says so with nulls rather than
|
|
130
|
+
// inventing a name — an unnamed run still gets timed, it just cannot be
|
|
131
|
+
// compared against a human baseline.
|
|
132
|
+
flowName = null,
|
|
133
|
+
minSteps = null,
|
|
115
134
|
options,
|
|
116
135
|
} = {},
|
|
117
136
|
) {
|
|
@@ -119,6 +138,24 @@ export async function runScript(
|
|
|
119
138
|
const { device } = await api.ensureDaemon(deviceQuery, options);
|
|
120
139
|
const udid = device.udid;
|
|
121
140
|
const startedAt = Date.now();
|
|
141
|
+
// Measurement only. Nothing below reads these, and a failure to write one
|
|
142
|
+
// can never change what a step does — see `note`.
|
|
143
|
+
const flowId = metrics.newFlowId();
|
|
144
|
+
const escalations = [];
|
|
145
|
+
const verdicts = [];
|
|
146
|
+
// Named noteEscalation, not note: the step loop below declares its own
|
|
147
|
+
// `note` string for the no-visible-change suffix, which shadowed this and
|
|
148
|
+
// turned every escalating verdict into a failed step reading "note is not a
|
|
149
|
+
// function". The try/catch inside here could not help — the throw was at the
|
|
150
|
+
// call site, one scope out. Instrumentation that can fail a flow is worse
|
|
151
|
+
// than no instrumentation.
|
|
152
|
+
const noteEscalation = (record) => {
|
|
153
|
+
try {
|
|
154
|
+
escalations.push(metrics.recordEscalation(udid, { flowId, flowName, ...record }));
|
|
155
|
+
} catch {
|
|
156
|
+
/* instrumentation must not be able to fail a flow it is only watching */
|
|
157
|
+
}
|
|
158
|
+
};
|
|
122
159
|
|
|
123
160
|
const needsInput = steps.some((s) => ACTION_STEPS.has(normalizeStep(s).action));
|
|
124
161
|
if (needsInput) {
|
|
@@ -145,10 +182,35 @@ export async function runScript(
|
|
|
145
182
|
// moves nothing and is not a failure, so an unbounded retry would rebuild the
|
|
146
183
|
// session and press again on every such step for no reason.
|
|
147
184
|
let inputRecovered = false;
|
|
185
|
+
/**
|
|
186
|
+
* The previous step's transition, still to be measured.
|
|
187
|
+
*
|
|
188
|
+
* Its pause profile cannot be read while the step is running — that is the
|
|
189
|
+
* biased measurement that corrupted the graph — so it is read one step later,
|
|
190
|
+
* off a frame history whose end nothing about the wait decided. See
|
|
191
|
+
* `api.longestQuietGap`.
|
|
192
|
+
*/
|
|
193
|
+
let pendingGap = null;
|
|
194
|
+
const measurePendingGap = async () => {
|
|
195
|
+
if (!pendingGap) return;
|
|
196
|
+
const { from, step: prevStep, actionAt } = pendingGap;
|
|
197
|
+
pendingGap = null;
|
|
198
|
+
try {
|
|
199
|
+
const history = (await api.getState(deviceQuery, { options })).state.history ?? [];
|
|
200
|
+
const trueGapMs = api.longestQuietGap(history, actionAt);
|
|
201
|
+
if (trueGapMs != null) graph.noteTrueGap(udid, from, prevStep, trueGapMs);
|
|
202
|
+
} catch {
|
|
203
|
+
/* a statistic nothing acts on must never be able to fail a flow */
|
|
204
|
+
}
|
|
205
|
+
};
|
|
148
206
|
|
|
149
207
|
for (const [i, raw] of steps.entries()) {
|
|
150
208
|
const step = normalizeStep(raw);
|
|
151
209
|
const stepStart = Date.now();
|
|
210
|
+
// Before anything else, and before this step disturbs the screen: the
|
|
211
|
+
// previous transition is definitely over by now, so its true pause profile
|
|
212
|
+
// is readable.
|
|
213
|
+
await measurePendingGap();
|
|
152
214
|
// The baseline for "did the screen react" must predate the action itself.
|
|
153
215
|
const beforeState = (await api.getState(deviceQuery, { options })).state;
|
|
154
216
|
const before = beforeState.hash;
|
|
@@ -164,14 +226,72 @@ export async function runScript(
|
|
|
164
226
|
// What this action did last time it was taken here, if ever.
|
|
165
227
|
const prediction = verify && beforeScreen?.hash ? graph.predict(udid, beforeScreen, step) : null;
|
|
166
228
|
try {
|
|
167
|
-
|
|
229
|
+
// How long this transition has cost before, on this screen, for this
|
|
230
|
+
// action. A cold edge gets the old fixed default and says so; a measured
|
|
231
|
+
// one gets p95 plus a margin. Research §7.
|
|
232
|
+
//
|
|
233
|
+
// Read before the step, not after, because one of the waits it informs
|
|
234
|
+
// happens *inside* the step: a `type into` taps the field and waits for
|
|
235
|
+
// focus before it types, and that wait used to be three constants.
|
|
236
|
+
const learned = verify && beforeScreen?.hash ? graph.timingFor(udid, beforeScreen, step) : null;
|
|
237
|
+
const focus = {
|
|
238
|
+
plan: graph.focusPlan(learned, {
|
|
239
|
+
reactionMs: FOCUS_REACTION_MS,
|
|
240
|
+
timeoutMs: FOCUS_TIMEOUT_MS,
|
|
241
|
+
stillnessMs: FOCUS_STABLE_MS,
|
|
242
|
+
keyboardUp: Boolean(beforeScreen?.keyboard),
|
|
243
|
+
}),
|
|
244
|
+
observedMs: null,
|
|
245
|
+
};
|
|
246
|
+
let detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
|
|
247
|
+
// How long this screen must hold still before it counts as settled.
|
|
248
|
+
//
|
|
249
|
+
// 500 ms was a constant paid by every step of every flow, and it is the
|
|
250
|
+
// reducible half of a settle: the rest is the transition genuinely
|
|
251
|
+
// taking time. An edge whose transition has never paused mid-flight
|
|
252
|
+
// needs 150 ms of quiet, not 500. Capped at the caller's own value, so
|
|
253
|
+
// this can only ever shorten a wait.
|
|
254
|
+
// NOT YET USED TO DECIDE ANYTHING, and the reason is worth the space.
|
|
255
|
+
//
|
|
256
|
+
// `graph.stillnessFor` computes a shorter window from the longest pause
|
|
257
|
+
// ever observed inside this transition, and measured live it made the
|
|
258
|
+
// Settings flow 8.0 s instead of 11.5 s — and wrong. Eight runs in a row
|
|
259
|
+
// failed at step 2 with the screen still showing Settings root, because
|
|
260
|
+
// step 1's settle returned mid-push, `screenIdentity` then read the
|
|
261
|
+
// screen we had not left yet, and the graph learned root -> root as a
|
|
262
|
+
// verified edge and started predicting it.
|
|
263
|
+
//
|
|
264
|
+
// The flaw is in the estimator, not the idea: the gap statistic is
|
|
265
|
+
// gathered only from what a wait itself observed, so a wait that ends
|
|
266
|
+
// early never sees the pauses that come later, the gaps look like zero,
|
|
267
|
+
// the window ratchets down, and the next wait ends earlier still. A
|
|
268
|
+
// self-reinforcing bias with a wrong graph at the end of it.
|
|
269
|
+
//
|
|
270
|
+
// The unbiased estimator is available and is a separate piece of work:
|
|
271
|
+
// the frame history holds every frame's timestamp and diff, so the true
|
|
272
|
+
// motion profile of a transition can be computed *after* it is over
|
|
273
|
+
// rather than from inside the wait that cut it short. Until then the
|
|
274
|
+
// gaps are recorded and not acted on — measuring is safe, and this is
|
|
275
|
+
// Phase 11's own rule that a learned number may only ever shorten a
|
|
276
|
+
// wait, applied to itself.
|
|
277
|
+
const stillness = step.stableMs ?? stableMs;
|
|
278
|
+
const stillnessPlan = learned ? graph.stillnessFor(learned, stillness) : { cold: true };
|
|
279
|
+
// A settle is not satisfied until the screen has held still for
|
|
280
|
+
// `stillness`, so a budget below that can never be met — and the learned
|
|
281
|
+
// p95 is measured from waits that include the stillness window, which
|
|
282
|
+
// makes it self-consistent but not self-evidently so. A tab switch with
|
|
283
|
+
// a p95 of 90ms would get a 240ms budget and then time out at 240ms
|
|
284
|
+
// waiting for 500ms of quiet, turning every fast edge into a failure.
|
|
285
|
+
const floorMs = stillness + 250;
|
|
286
|
+
const budgetMs = step.timeoutMs
|
|
287
|
+
?? (learned && !learned.cold ? Math.max(learned.timeoutMs, floorMs) : timeoutMs);
|
|
168
288
|
const settleFor = async () => {
|
|
169
289
|
if (!autoSettle || !ACTION_STEPS.has(step.action)) return null;
|
|
170
290
|
const w = await api.waitFor(deviceQuery, {
|
|
171
291
|
mode: 'settle',
|
|
172
292
|
since: before,
|
|
173
|
-
stableMs:
|
|
174
|
-
timeoutMs:
|
|
293
|
+
stableMs: stillness,
|
|
294
|
+
timeoutMs: budgetMs,
|
|
175
295
|
options,
|
|
176
296
|
});
|
|
177
297
|
return {
|
|
@@ -180,9 +300,64 @@ export async function runScript(
|
|
|
180
300
|
sawChange: w.sawChange,
|
|
181
301
|
stalled: Boolean(w.stalled),
|
|
182
302
|
noVisibleChange: Boolean(w.noVisibleChange),
|
|
303
|
+
// What this wait was allowed, and where the number came from. A
|
|
304
|
+
// timeout nobody can explain is how a fixed sleep comes back as a
|
|
305
|
+
// constant with a comment.
|
|
306
|
+
budgetMs,
|
|
307
|
+
stillnessMs: stillness,
|
|
308
|
+
quietGapMs: w.quietGapMs,
|
|
309
|
+
// The baseline had already finished moving when the wait began, so it
|
|
310
|
+
// was re-taken from the live screen. Surfaced because it means the
|
|
311
|
+
// step before this one had not finished when this one started.
|
|
312
|
+
staleBaseline: Boolean(w.staleBaseline),
|
|
313
|
+
// Something moved in one region only — a switch, a radio dot, a
|
|
314
|
+
// segment highlight. Worth saying, because it is the difference
|
|
315
|
+
// between "the action did nothing" and "the action did something the
|
|
316
|
+
// whole-screen mean cannot see".
|
|
317
|
+
smallChange: Boolean(w.smallChange),
|
|
318
|
+
timing: learned
|
|
319
|
+
? {
|
|
320
|
+
p50: learned.p50,
|
|
321
|
+
p95: learned.p95,
|
|
322
|
+
samples: learned.samples,
|
|
323
|
+
cold: learned.cold,
|
|
324
|
+
gapP95: learned.gapP95,
|
|
325
|
+
gapSamples: learned.gapSamples,
|
|
326
|
+
// What it *would* have been, for the eval that has to happen
|
|
327
|
+
// before this is trusted with a wait.
|
|
328
|
+
stillnessWouldBe: stillnessPlan.stillnessMs ?? null,
|
|
329
|
+
}
|
|
330
|
+
: null,
|
|
183
331
|
};
|
|
184
332
|
};
|
|
185
333
|
let settled = await settleFor();
|
|
334
|
+
// A screen that is still working earns more time; a screen doing nothing
|
|
335
|
+
// visible has already answered. Research §7: keep waiting past p95 only
|
|
336
|
+
// while the transition classifier says something is loading, and never
|
|
337
|
+
// past Nielsen's 10 s — at which point it escalates with the timing
|
|
338
|
+
// attached rather than waiting longer.
|
|
339
|
+
if (settled && !settled.ok && !settled.noVisibleChange && learned && !learned.cold) {
|
|
340
|
+
const kind = (await api.getState(deviceQuery, { options })).state.transition?.kind;
|
|
341
|
+
const verdict = graph.slowerThanUsual({
|
|
342
|
+
elapsedMs: settled.waitedMs, p95: learned.p95, settled: false, kind,
|
|
343
|
+
});
|
|
344
|
+
if (verdict.keepWaiting) {
|
|
345
|
+
const remaining = graph.HARD_CAP_MS - settled.waitedMs;
|
|
346
|
+
const more = await api.waitFor(deviceQuery, {
|
|
347
|
+
mode: 'settle', since: before, stableMs: stillness, timeoutMs: remaining, options,
|
|
348
|
+
});
|
|
349
|
+
settled = {
|
|
350
|
+
...settled,
|
|
351
|
+
ok: more.satisfied,
|
|
352
|
+
waitedMs: settled.waitedMs + more.waitedMs,
|
|
353
|
+
sawChange: settled.sawChange || more.sawChange,
|
|
354
|
+
quietGapMs: Math.max(settled.quietGapMs ?? 0, more.quietGapMs ?? 0),
|
|
355
|
+
slowerThanUsual: verdict.note,
|
|
356
|
+
};
|
|
357
|
+
} else if (verdict.slower) {
|
|
358
|
+
settled = { ...settled, slowerThanUsual: verdict.note };
|
|
359
|
+
}
|
|
360
|
+
}
|
|
186
361
|
|
|
187
362
|
// A hardware button that moved nothing did not arrive.
|
|
188
363
|
//
|
|
@@ -226,14 +401,66 @@ export async function runScript(
|
|
|
226
401
|
// for. Requiring both meant a screen that settled slowly recorded
|
|
227
402
|
// nothing at all.
|
|
228
403
|
endScreen = afterScreen;
|
|
229
|
-
|
|
230
|
-
|
|
404
|
+
// An action with no observed effect teaches the graph nothing, and
|
|
405
|
+
// recording it teaches something false.
|
|
406
|
+
//
|
|
407
|
+
// This is the second half of the same bug. A settle that returned on a
|
|
408
|
+
// stale baseline reported `ok` for a tap that moved nothing, and the
|
|
409
|
+
// recorder asked only whether the *reading* was confirmed — so
|
|
410
|
+
// `root -> root` went in as a verified edge and started being
|
|
411
|
+
// predicted. Re-baselining stops the settle lying; this stops the
|
|
412
|
+
// graph learning from a step that has no evidence behind it either way.
|
|
413
|
+
//
|
|
414
|
+
// It does cost a real case for now: a control that genuinely returns to
|
|
415
|
+
// the same screen — a toggle — is invisible to the change detector at
|
|
416
|
+
// eight times below its threshold, so it reads as no-visible-change and
|
|
417
|
+
// its edge is no longer recorded. That is the right trade while the
|
|
418
|
+
// detector cannot see it, and it comes back on its own once it can.
|
|
419
|
+
const noEvidence = Boolean(settled?.noVisibleChange);
|
|
420
|
+
if (afterScreen.confirmed && afterScreen.hash && !noEvidence) {
|
|
421
|
+
// The observed cost of this transition, which is what makes the next
|
|
422
|
+
// one adaptive. Only from a settle that was actually satisfied: a
|
|
423
|
+
// timeout is not a measurement of how long the screen takes, it is a
|
|
424
|
+
// measurement of how long we were prepared to wait.
|
|
425
|
+
graph.record(udid, {
|
|
426
|
+
from: beforeScreen, action: step, to: afterScreen, kind,
|
|
427
|
+
settleMs: settled?.ok ? settled.waitedMs : undefined,
|
|
428
|
+
// The pause statistic is worth having from any settle that saw the
|
|
429
|
+
// screen move, satisfied or not: a transition that paused for
|
|
430
|
+
// 400ms and then timed out is exactly the case a 150ms stillness
|
|
431
|
+
// window would have got wrong.
|
|
432
|
+
quietGapMs: settled?.sawChange ? settled.quietGapMs : undefined,
|
|
433
|
+
// The focus wait's own distribution, kept apart from the step's.
|
|
434
|
+
// Only set when a field was tapped and visibly took focus.
|
|
435
|
+
focusMs: focus.observedMs ?? undefined,
|
|
436
|
+
});
|
|
437
|
+
pendingGap = { from: beforeScreen, step, actionAt: stepStart };
|
|
438
|
+
carriedScreen = afterScreen;
|
|
439
|
+
} else if (afterScreen.confirmed && afterScreen.hash) {
|
|
440
|
+
// Where we are is still known; only what got us here is not worth
|
|
441
|
+
// remembering. Carrying it saves the next step a perception pass.
|
|
231
442
|
carriedScreen = afterScreen;
|
|
232
443
|
}
|
|
233
444
|
}
|
|
234
445
|
|
|
235
446
|
const wrongTurn = wrongTurnFrom(verification);
|
|
236
|
-
|
|
447
|
+
// `[no visible change]` after a launch is ambiguous between two very
|
|
448
|
+
// different things, and a real session read it the wrong way twice:
|
|
449
|
+
// "the app was already in front, so nothing needed to move" and "the app
|
|
450
|
+
// did not come forward". Measured on this Xcode, `simctl launch` *does*
|
|
451
|
+
// front an already-running app, so the first reading is the likely one —
|
|
452
|
+
// but likely is not the same as said, and the step is the only place that
|
|
453
|
+
// can say it.
|
|
454
|
+
const launchNote = step.action === 'launch' && settled?.noVisibleChange
|
|
455
|
+
? ' [the screen did not change, so this app was already in front — or it did not come forward]'
|
|
456
|
+
: '';
|
|
457
|
+
const note = launchNote
|
|
458
|
+
+ (settled?.smallChange ? ' [a small change, in one region only]' : '')
|
|
459
|
+
+ (settled?.noVisibleChange ? ' [no visible change]' : '')
|
|
460
|
+
+ (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
|
|
461
|
+
+ (settled?.blackFrames
|
|
462
|
+
? ` [${settled.blackFrames} black frame(s) waited through${settled.blackMs ? `, still black after ${settled.blackMs}ms` : ''}]`
|
|
463
|
+
: '');
|
|
237
464
|
results.push({
|
|
238
465
|
index: i,
|
|
239
466
|
action: step.action,
|
|
@@ -244,6 +471,25 @@ export async function runScript(
|
|
|
244
471
|
settled,
|
|
245
472
|
});
|
|
246
473
|
const halt = haltDecision({ verification, stopOnUnexpected, continueOnError });
|
|
474
|
+
if (verification?.verdict) verdicts.push(verification.verdict);
|
|
475
|
+
if (metrics.ESCALATING_VERDICTS.has(verification?.verdict)) {
|
|
476
|
+
noteEscalation({
|
|
477
|
+
stepIndex: i,
|
|
478
|
+
fingerprint: beforeScreen?.hash ?? null,
|
|
479
|
+
reason: 'verification_failed',
|
|
480
|
+
candidates: [],
|
|
481
|
+
// A halted run is a decision simframe made and stopped on; a step
|
|
482
|
+
// that moved nothing carries on and leaves the judgement to whoever
|
|
483
|
+
// reads the result.
|
|
484
|
+
outcome: halt.halt ? 'failed' : 'escalated_to_model',
|
|
485
|
+
wallMs: Date.now() - stepStart,
|
|
486
|
+
detail: `${verification.verdict}: ${verification.detail}`
|
|
487
|
+
+ (settled?.slowerThanUsual ? ` [${settled.slowerThanUsual}]` : '')
|
|
488
|
+
+ (settled?.timing && !settled.timing.cold
|
|
489
|
+
? ` [waited ${settled.waitedMs}ms of a ${settled.budgetMs}ms budget; p95 ${settled.timing.p95}ms]`
|
|
490
|
+
: ''),
|
|
491
|
+
});
|
|
492
|
+
}
|
|
247
493
|
if (halt.halt) {
|
|
248
494
|
results[results.length - 1].ok = false;
|
|
249
495
|
results[results.length - 1].error = halt.error;
|
|
@@ -252,20 +498,56 @@ export async function runScript(
|
|
|
252
498
|
}
|
|
253
499
|
} catch (err) {
|
|
254
500
|
results.push({ index: i, action: step.action, ok: false, ms: Date.now() - stepStart, error: err.message });
|
|
501
|
+
const why = metrics.reasonForStepError(step, err);
|
|
502
|
+
noteEscalation({
|
|
503
|
+
stepIndex: i,
|
|
504
|
+
fingerprint: beforeScreen?.hash ?? metrics.fingerprintNow(udid, screenmap),
|
|
505
|
+
reason: why.reason,
|
|
506
|
+
candidates: why.candidates,
|
|
507
|
+
tried: why.tried,
|
|
508
|
+
outcome: 'failed',
|
|
509
|
+
wallMs: Date.now() - stepStart,
|
|
510
|
+
detail: err.message,
|
|
511
|
+
});
|
|
255
512
|
failed = true;
|
|
256
513
|
if (!continueOnError) break;
|
|
257
514
|
}
|
|
258
515
|
}
|
|
259
516
|
|
|
517
|
+
// The last step has no next step to measure it, and its transition is over by
|
|
518
|
+
// the time the loop exits.
|
|
519
|
+
await measurePendingGap();
|
|
520
|
+
|
|
521
|
+
const wallMs = Date.now() - startedAt;
|
|
522
|
+
try {
|
|
523
|
+
metrics.recordFlow(udid, metrics.flowRecordFrom({
|
|
524
|
+
flowId,
|
|
525
|
+
flowName,
|
|
526
|
+
udid,
|
|
527
|
+
startedAt,
|
|
528
|
+
wallMs,
|
|
529
|
+
stepsTaken: results.length,
|
|
530
|
+
totalSteps: steps.length,
|
|
531
|
+
minSteps,
|
|
532
|
+
imagesSent: frames.length,
|
|
533
|
+
escalations,
|
|
534
|
+
verdicts,
|
|
535
|
+
completed: !failed && results.length === steps.length,
|
|
536
|
+
}));
|
|
537
|
+
} catch {
|
|
538
|
+
/* as above: a flow that ran is not a flow that failed because of a log */
|
|
539
|
+
}
|
|
540
|
+
|
|
260
541
|
return {
|
|
261
542
|
device,
|
|
262
543
|
// Returned so a run that verified end to end can be handed straight to
|
|
263
544
|
// navigate.saveFlow without the caller reassembling what it just ran.
|
|
264
545
|
steps,
|
|
546
|
+
flowId,
|
|
265
547
|
endScreen,
|
|
266
548
|
results,
|
|
267
549
|
ok: !failed,
|
|
268
|
-
totalMs:
|
|
550
|
+
totalMs: wallMs,
|
|
269
551
|
ranSteps: results.length,
|
|
270
552
|
totalSteps: steps.length,
|
|
271
553
|
frames,
|
|
@@ -284,18 +566,31 @@ export async function runScript(
|
|
|
284
566
|
*/
|
|
285
567
|
async function focusField(deviceQuery, udid, step, ctx) {
|
|
286
568
|
const found = await api.locate(deviceQuery, step.into, { index: step.index, refresh: step.refresh });
|
|
569
|
+
// What this field has cost to focus before, on this screen. Cold, or with no
|
|
570
|
+
// verification running, that is exactly the three constants above; measured,
|
|
571
|
+
// it can only be longer. `graph.focusPlan` carries the reason it is either.
|
|
572
|
+
const plan = ctx.focus?.plan ?? {
|
|
573
|
+
reactionMs: FOCUS_REACTION_MS, timeoutMs: FOCUS_TIMEOUT_MS, cold: true, from: 'no timing in hand',
|
|
574
|
+
};
|
|
575
|
+
const tappedAt = Date.now();
|
|
287
576
|
await input.tapPoint(udid, found.target.x, found.target.y);
|
|
288
577
|
const focused = await api.waitFor(deviceQuery, {
|
|
289
578
|
mode: 'settle',
|
|
290
579
|
stableMs: FOCUS_STABLE_MS,
|
|
291
|
-
reactionMs:
|
|
292
|
-
timeoutMs:
|
|
580
|
+
reactionMs: plan.reactionMs,
|
|
581
|
+
timeoutMs: plan.timeoutMs,
|
|
293
582
|
options: ctx.options,
|
|
294
583
|
});
|
|
584
|
+
// Only a wait that was satisfied is a measurement of how long focus takes. A
|
|
585
|
+
// reaction window that ran out measures how long we were prepared to watch a
|
|
586
|
+
// screen that did not move, and banking that would teach the edge the cost of
|
|
587
|
+
// its own impatience — the estimator mistake learned stillness made.
|
|
588
|
+
if (ctx.focus && focused.satisfied) ctx.focus.observedMs = Date.now() - tappedAt;
|
|
295
589
|
return {
|
|
296
590
|
found,
|
|
297
591
|
where: `"${found.target.label}" at ${found.target.x},${found.target.y}`,
|
|
298
592
|
quiet: focused.satisfied ? '' : ' [the field did not visibly take focus]',
|
|
593
|
+
waited: focused.satisfied && !plan.cold ? ` [focus in ${focused.waitedMs}ms, ${plan.from}]` : '',
|
|
299
594
|
};
|
|
300
595
|
}
|
|
301
596
|
|
|
@@ -327,7 +622,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
327
622
|
if (step.into) {
|
|
328
623
|
const field = await focusField(deviceQuery, udid, step, ctx);
|
|
329
624
|
await input.typeText(udid, step.text ?? step.value);
|
|
330
|
-
return `typed into ${field.where}${field.quiet}`;
|
|
625
|
+
return `typed into ${field.where}${field.quiet}${field.waited}`;
|
|
331
626
|
}
|
|
332
627
|
await input.typeText(udid, step.text ?? step.value);
|
|
333
628
|
return 'typed text';
|
|
@@ -340,7 +635,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
340
635
|
if (step.into) {
|
|
341
636
|
const field = await focusField(deviceQuery, udid, step, ctx);
|
|
342
637
|
await input.pasteText(udid, step.text ?? step.value);
|
|
343
|
-
return `pasted into ${field.where}${field.quiet}`;
|
|
638
|
+
return `pasted into ${field.where}${field.quiet}${field.waited}`;
|
|
344
639
|
}
|
|
345
640
|
await input.pasteText(udid, step.text ?? step.value);
|
|
346
641
|
return 'pasted into the focused field';
|
|
@@ -420,6 +715,14 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
420
715
|
return `"${node.label ?? target}" appeared`;
|
|
421
716
|
} catch (err) {
|
|
422
717
|
lastError = err.message;
|
|
718
|
+
// Same rule as `waitFor`, and `matchElement` says it in its own
|
|
719
|
+
// words: a query that matched several elements has found them all
|
|
720
|
+
// already.
|
|
721
|
+
if (/matched \d+ elements/.test(err.message)) {
|
|
722
|
+
throw new Error(
|
|
723
|
+
`${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
|
|
724
|
+
);
|
|
725
|
+
}
|
|
423
726
|
}
|
|
424
727
|
await sleep(250);
|
|
425
728
|
}
|
|
@@ -472,6 +775,22 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
472
775
|
return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`;
|
|
473
776
|
} catch (err) {
|
|
474
777
|
lastError = err.message;
|
|
778
|
+
// Waiting cannot make a thing unique.
|
|
779
|
+
//
|
|
780
|
+
// Reported from a real session: a wait on an ambiguous string spent
|
|
781
|
+
// the full 30 s and then listed four matches, all four of which were
|
|
782
|
+
// on the very first frame. The disambiguation is good and it arrived
|
|
783
|
+
// twenty-nine seconds after everything it needed. `ambiguous` means
|
|
784
|
+
// the target is *present*, several times over — which is precisely
|
|
785
|
+
// the case where more time changes nothing.
|
|
786
|
+
//
|
|
787
|
+
// Distinguished by the tag at the throw site rather than by reading
|
|
788
|
+
// the message, because "not on this screen" tags the same reason.
|
|
789
|
+
if (metrics.escalationOf(err)?.ambiguous) {
|
|
790
|
+
throw new Error(
|
|
791
|
+
`${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
|
|
792
|
+
);
|
|
793
|
+
}
|
|
475
794
|
}
|
|
476
795
|
if (Date.now() >= limit) break;
|
|
477
796
|
await sleep(POLL_MS);
|
package/src/analyze.js
CHANGED
|
@@ -35,6 +35,42 @@ export function signatureDiff(a, b) {
|
|
|
35
35
|
return sum / a.length / 255;
|
|
36
36
|
}
|
|
37
37
|
|
|
38
|
+
/**
|
|
39
|
+
* A change small enough that the whole-screen mean cannot see it.
|
|
40
|
+
*
|
|
41
|
+
* `changed` in both daemons is `signatureDiff > 0.004`, a mean over a 4x8 grid
|
|
42
|
+
* of gray means. Measured on this device, an iOS switch flipping:
|
|
43
|
+
*
|
|
44
|
+
* mean diff 0.001348 — a third of the threshold, so: not a change
|
|
45
|
+
* max cell delta 0.043137 — one cell of thirty-two, in row 1
|
|
46
|
+
*
|
|
47
|
+
* So the entire class of small binary controls — switches, radio dots,
|
|
48
|
+
* checkboxes, segment highlights — changes nothing as far as the daemon is
|
|
49
|
+
* concerned, and a step that flips one reports `no-visible-change`, which is a
|
|
50
|
+
* verdict that escalates.
|
|
51
|
+
*
|
|
52
|
+
* The threshold sits in a measured gap rather than being chosen. Eighty seconds
|
|
53
|
+
* of a *static* screen gave a largest per-cell delta of 0.003922, in row 0,
|
|
54
|
+
* which is the status-bar clock ticking over — the only thing moving. So the
|
|
55
|
+
* separation is 0.0039 against 0.0431, eleven times, and 0.012 is three times
|
|
56
|
+
* the noise and three and a half times under the signal. No row is excluded:
|
|
57
|
+
* the clock does not reach the threshold, which is a better reason to ignore it
|
|
58
|
+
* than a structural exclusion that would also blind the nav bar.
|
|
59
|
+
*
|
|
60
|
+
* What this deliberately does **not** do is feed stillness. `stableForMs` stays
|
|
61
|
+
* on the mean, because a blinking text caret is a small localised change and a
|
|
62
|
+
* screen with a cursor in it would otherwise never settle. The two signals are
|
|
63
|
+
* independent by design: this one answers "did the action do anything", and the
|
|
64
|
+
* mean answers "has the screen finished moving".
|
|
65
|
+
*/
|
|
66
|
+
export const CELL_CHANGE = 0.012;
|
|
67
|
+
|
|
68
|
+
/** The largest single-region change between two signatures, 0-1. */
|
|
69
|
+
export function maxCellDelta(a, b) {
|
|
70
|
+
const deltas = regionDeltas(a, b);
|
|
71
|
+
return deltas.length ? Math.max(...deltas) : 0;
|
|
72
|
+
}
|
|
73
|
+
|
|
38
74
|
/** Per-region change fractions, so callers can tell a toast from a screen push. */
|
|
39
75
|
export function regionDeltas(a, b) {
|
|
40
76
|
if (!a || !b || a.length !== b.length) return a ? a.map(() => 1) : [];
|
|
@@ -69,6 +105,40 @@ export function signatureToHex(sig) {
|
|
|
69
105
|
return sig.map((v) => v.toString(16).padStart(2, '0')).join('');
|
|
70
106
|
}
|
|
71
107
|
|
|
108
|
+
/**
|
|
109
|
+
* Is this frame black — not dark, black?
|
|
110
|
+
*
|
|
111
|
+
* The capture wedge (docs/BENCHMARKS.md) leaves `simctl io screenshot`
|
|
112
|
+
* succeeding and returning 0 non-black pixels of 3,162,132, and simframe's own
|
|
113
|
+
* frames go the same way: the display pipeline stops rendering while
|
|
114
|
+
* everything about the capture path keeps reporting success. It self-recovers
|
|
115
|
+
* most times and a device restart cures the rest, so the useful thing is to
|
|
116
|
+
* notice early — before a settle spends its whole budget deciding a black
|
|
117
|
+
* screen is a calm one.
|
|
118
|
+
*
|
|
119
|
+
* The signature is already computed for every frame, so this costs 32 integer
|
|
120
|
+
* comparisons and no decode. A real screen does not come close: measured on
|
|
121
|
+
* this device's Settings root, the 32 bytes ran 191–245.
|
|
122
|
+
*
|
|
123
|
+
* The threshold is a level, not a fraction, and it is on the *maximum*: one
|
|
124
|
+
* cell with anything in it is enough to say the display is rendering. That
|
|
125
|
+
* matters because a dark-mode screen, a video, or a splash on black are all
|
|
126
|
+
* legitimately near-zero in most cells and this must not call them faults.
|
|
127
|
+
*
|
|
128
|
+
* And it says "the frames are black", never "the simulator is wedged". A
|
|
129
|
+
* screen can be black because the app drew black. What makes it a wedge is
|
|
130
|
+
* that it stays black while input is being delivered, and only the caller
|
|
131
|
+
* knows that.
|
|
132
|
+
*/
|
|
133
|
+
export const BLACK_LEVEL = 8;
|
|
134
|
+
|
|
135
|
+
export function isBlackFrame(sig, { level = BLACK_LEVEL } = {}) {
|
|
136
|
+
const bytes = typeof sig === 'string' ? hexToSignature(sig) : sig;
|
|
137
|
+
if (!bytes?.length) return false;
|
|
138
|
+
for (const b of bytes) if (b > level) return false;
|
|
139
|
+
return true;
|
|
140
|
+
}
|
|
141
|
+
|
|
72
142
|
export function hexToSignature(hex) {
|
|
73
143
|
const out = [];
|
|
74
144
|
for (let i = 0; i < hex.length; i += 2) out.push(parseInt(hex.slice(i, i + 2), 16));
|