simframe 0.17.0 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +126 -1058
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +12 -6
- package/package.json +1 -1
- package/scripts/article-md.mjs +111 -45
- package/scripts/bench-hpi.mjs +53 -1
- package/scripts/ci-device-guard.mjs +27 -0
- package/scripts/ci-memory.mjs +33 -6
- package/scripts/demo-gif/README.md +36 -0
- package/scripts/demo-gif/compose.swift +106 -0
- package/scripts/demo-gif/events.example.json +74 -0
- package/scripts/demo-gif/flow.json +6 -0
- package/scripts/device-state.mjs +5 -62
- package/scripts/smithery/icon.png +0 -0
- package/scripts/smithery-bundle.mjs +77 -0
- package/src/actions.js +309 -32
- package/src/cli.js +111 -29
- package/src/device-state.js +101 -0
- package/src/index.js +139 -11
- package/src/metrics.js +144 -3
- package/src/navigate.js +10 -0
- package/src/platform/cdp.js +242 -0
- package/src/platform/index.js +28 -2
- package/src/store.js +39 -0
- package/src/wedge.js +370 -0
package/src/cli.js
CHANGED
|
@@ -3,7 +3,7 @@ import fs from 'node:fs';
|
|
|
3
3
|
import os from 'node:os';
|
|
4
4
|
import path from 'node:path';
|
|
5
5
|
import { runDaemon, DEFAULTS } from './daemon.js';
|
|
6
|
-
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice,
|
|
6
|
+
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
|
|
7
7
|
import * as actions from './actions.js';
|
|
8
8
|
import * as analyze from './analyze.js';
|
|
9
9
|
import * as api from './index.js';
|
|
@@ -11,11 +11,15 @@ import * as input from './input.js';
|
|
|
11
11
|
import * as baseline from './baseline.js';
|
|
12
12
|
import * as metrics from './metrics.js';
|
|
13
13
|
import * as navigate from './navigate.js';
|
|
14
|
+
import * as wedge from './wedge.js';
|
|
15
|
+
|
|
14
16
|
import { decodePng } from './png.js';
|
|
15
17
|
import * as storage from './storage.js';
|
|
16
18
|
import * as store from './store.js';
|
|
17
19
|
import * as view from './view.js';
|
|
18
20
|
|
|
21
|
+
|
|
22
|
+
|
|
19
23
|
const USAGE = `simframe — always-warm iOS Simulator frames
|
|
20
24
|
|
|
21
25
|
simframe mcp run the MCP server on stdio (for agents)
|
|
@@ -51,6 +55,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
51
55
|
simframe escalations [device] why simframe handed decisions back, by reason
|
|
52
56
|
simframe supervisions [device] local supervisor rulings, and what came of each
|
|
53
57
|
simframe revive [device] power-cycle a wedged device: stop, shutdown, boot, start, reset input
|
|
58
|
+
simframe diagnose [device] what this device is doing right now, and which failure it is
|
|
54
59
|
(--session=<id> narrows to one agent; the
|
|
55
60
|
ids are listed in the output. SIMFRAME_SESSION
|
|
56
61
|
names one, but only at process start — an
|
|
@@ -419,31 +424,67 @@ async function main() {
|
|
|
419
424
|
case 'revive': {
|
|
420
425
|
const dev = await resolveDevice(device);
|
|
421
426
|
const say = (line) => { if (!flags.json) console.log(line); };
|
|
422
|
-
const steps = [];
|
|
423
|
-
const did = async (what, fn) => {
|
|
424
|
-
try { await fn(); steps.push({ step: what, ok: true }); say(` ok ${what}`); } catch (err) {
|
|
425
|
-
steps.push({ step: what, ok: false, error: err.message });
|
|
426
|
-
say(` .. ${what} — ${err.message.split('\n')[0]}`);
|
|
427
|
-
}
|
|
428
|
-
};
|
|
429
427
|
say(`reviving ${dev.name}`);
|
|
430
|
-
//
|
|
431
|
-
//
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
const
|
|
441
|
-
const
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
428
|
+
// The sequence itself lives in `wedge.js`, because `bench-hpi` needs the
|
|
429
|
+
// same recovery between passes and a second copy of it here is how this
|
|
430
|
+
// project has repeatedly ended up fixing one symptom in three places.
|
|
431
|
+
const revived = await wedge.revive(dev.udid, {
|
|
432
|
+
options,
|
|
433
|
+
device: dev,
|
|
434
|
+
onStep: ({ step, ok, error }) => say(ok ? ` ok ${step}` : ` .. ${step} — ${String(error).split('\n')[0]}`),
|
|
435
|
+
});
|
|
436
|
+
const { steps } = revived;
|
|
437
|
+
const diag = { verdict: revived.verdict };
|
|
438
|
+
const state = diag?.verdict?.state ?? 'read-failed';
|
|
439
|
+
const usable = !wedge.UNUSABLE.has(state);
|
|
440
|
+
const degraded = usable && state !== 'healthy';
|
|
441
|
+
emit(flags, { ok: usable, device: dev.udid, steps, verdict: diag?.verdict ?? null }, usable
|
|
442
|
+
? (degraded
|
|
443
|
+
? `\n${dev.name} is back and usable, but ${state}:\n ${diag.verdict.detail}`
|
|
444
|
+
+ '\nNot a failure — it taps and reads.'
|
|
445
|
+
: `\n${dev.name} is healthy again — ${diag.verdict.detail}`)
|
|
446
|
+
: `\n${dev.name} came back ${state}, which is not usable.`
|
|
447
|
+
+ `\n ${diag?.verdict?.detail ?? 'nothing could be read from it'}`
|
|
448
|
+
+ '\nA second revive sometimes clears it. If it does not, this is past what simframe'
|
|
449
|
+
+ ' can do — check Simulator.app is not showing an error, and see docs/DEFERRED.md'
|
|
450
|
+
+ ' items 95 and 183.');
|
|
451
|
+
if (!usable) process.exitCode = 1;
|
|
452
|
+
return;
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
// Not folded into `doctor`, which answers "can this machine capture". This
|
|
456
|
+
// answers "what is this device doing right now", which is item 173's
|
|
457
|
+
// question and has never had an instrument.
|
|
458
|
+
case 'diagnose': {
|
|
459
|
+
const dev = await resolveDevice(device);
|
|
460
|
+
const r = await wedge.diagnose(dev.udid, { options });
|
|
461
|
+
emit(flags, r, [
|
|
462
|
+
`${r.device.name} — ${r.verdict.state}`,
|
|
463
|
+
` ${r.verdict.detail}`,
|
|
464
|
+
'',
|
|
465
|
+
` frame seq ${r.frame?.seq ?? '-'}, ${r.frame?.ageMs ?? '-'}ms old, still for ${r.frame?.stableForMs ?? '-'}ms, ${r.frame?.size ?? '-'}`,
|
|
466
|
+
` elements ${r.elements.total} total — ${r.elements.ax} by tree, ${r.elements.ocr} by OCR, ${r.elements.fused} by both`,
|
|
467
|
+
// **No band here, and that is the second version of this fix.**
|
|
468
|
+
//
|
|
469
|
+
// A reporter saw `healthy` printed beside "measured healthy 0.66-0.93"
|
|
470
|
+
// over a reading of 0.567 and asked, reasonably, which to believe. The
|
|
471
|
+
// first fix widened the band to 0.57-0.93 — and the very next device
|
|
472
|
+
// read returned **0.471** on an ordinary Settings screen, which is the
|
|
473
|
+
// same contradiction one decimal place down. Any fixed band will be
|
|
474
|
+
// contradicted by the next screen, because fusion tracks how much of a
|
|
475
|
+
// screen's content is OCR-only and that is a property of the app.
|
|
476
|
+
//
|
|
477
|
+
// So the line prints the number and the one threshold the verdict
|
|
478
|
+
// actually turns on. The observed range lives in `docs/BENCHMARKS.md`,
|
|
479
|
+
// where it is evidence about screens rather than a standard a device is
|
|
480
|
+
// being held to. `HEALTHY_FUSION` is still printed by the `stale-frame`
|
|
481
|
+
// verdict, where a reading of 0.03 genuinely wants the contrast.
|
|
482
|
+
` fusion ${r.agreement ?? 'n/a'} of elements seen by both sensors`
|
|
483
|
+
+ ` — stale at or below ${wedge.DISAGREEMENT}, which is what this verdict turns on`,
|
|
484
|
+
` frontmost ${r.frontmost?.pid ?? 'unknown'}${r.frontmost?.title ? ` (${r.frontmost.title})` : ''}`,
|
|
485
|
+
...(r.verdict.revive ? ['', ' `simframe revive` is the recovery. Keep this output — item 173 needs it.'] : []),
|
|
486
|
+
]);
|
|
487
|
+
if (r.verdict.revive) process.exitCode = 1;
|
|
447
488
|
return;
|
|
448
489
|
}
|
|
449
490
|
|
|
@@ -1253,8 +1294,14 @@ async function main() {
|
|
|
1253
1294
|
'flow runs agent p50 human p50 HPI_time step_ratio turns esc',
|
|
1254
1295
|
...report.flows.map((f) => metrics.flowRow(f, { wide: true })),
|
|
1255
1296
|
'',
|
|
1256
|
-
`HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs}
|
|
1297
|
+
`HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs} measurable run(s), ` +
|
|
1257
1298
|
`${report.overall.runs - runs.filter((r) => r.completed && !r.wrong_action_taken).length} not clean)`,
|
|
1299
|
+
// A denominator that quietly shrinks is worse than one that is wrong.
|
|
1300
|
+
report.overall.runs_lost_to_device
|
|
1301
|
+
? ` ${report.overall.runs_lost_to_device} further run(s) left the denominator because the DEVICE failed,`
|
|
1302
|
+
+ ' not the code — those are unmeasured, not inaccurate (item 173):'
|
|
1303
|
+
+ `\n${(report.overall.device_causes ?? []).map((c) => ` ${c}`).join('\n')}`
|
|
1304
|
+
: null,
|
|
1258
1305
|
report.overall.hpi_time == null
|
|
1259
1306
|
? `HPI_time and HPI need a human baseline — none of ${report.overall.flows_measured} measured flow(s) has one yet.`
|
|
1260
1307
|
: `HPI_time ${report.overall.hpi_time} (harmonic mean over ${report.overall.flows_with_human_baseline} flow(s)), HPI ${report.overall.hpi}`,
|
|
@@ -1265,7 +1312,10 @@ async function main() {
|
|
|
1265
1312
|
`steps per model call ${report.overall.steps_per_call ?? '—'}`
|
|
1266
1313
|
+ (report.overall.steps_per_call
|
|
1267
1314
|
? ` — about ${(((20000 + 1700 * report.overall.steps_per_call) / report.overall.steps_per_call) / 1000).toFixed(1)}s`
|
|
1268
|
-
+ ' per step end to end
|
|
1315
|
+
+ ' per step end to end. Raise this, not the engine — but note the ~1.7s'
|
|
1316
|
+
+ ' figure for simframe\'s own work is WARM taps inside a batch: a cold app'
|
|
1317
|
+
+ ' launch is a large one-off on top, and it dominates short flows'
|
|
1318
|
+
+ ' (measured: 6.2s/step over 2 steps with no model at all).'
|
|
1269
1319
|
: ''),
|
|
1270
1320
|
flags.out ? `wrote ${flags.out}` : null,
|
|
1271
1321
|
]);
|
|
@@ -1354,13 +1404,45 @@ async function main() {
|
|
|
1354
1404
|
if (read > 0) {
|
|
1355
1405
|
verdict = named + (read < n ? ` (on the ${read} of ${n} whose reason was read)` : '');
|
|
1356
1406
|
} else if (assumed > 0) {
|
|
1357
|
-
|
|
1407
|
+
// For `verification_failed` this line used to end the story, and
|
|
1408
|
+
// it is no longer the whole truth: the verdict split below names
|
|
1409
|
+
// a faculty for some of them and takes the device's own failures
|
|
1410
|
+
// out altogether.
|
|
1411
|
+
verdict = r === 'verification_failed' && (b.verdicts?.length || b.device_total)
|
|
1412
|
+
? 'reason too coarse to steer by — see the verdict split below'
|
|
1413
|
+
: 'reason assumed, not read — no faculty can be named from these';
|
|
1358
1414
|
} else {
|
|
1359
1415
|
// Neither read nor assumed: the log predates the distinction.
|
|
1360
|
-
verdict = `${named} — but these records predate the check, so treat it as untested
|
|
1416
|
+
verdict = `${named} — but these records predate the check, so treat it as untested`
|
|
1417
|
+
+ ' (legacy, not assumed)';
|
|
1361
1418
|
}
|
|
1362
1419
|
return ` ${r.padEnd(20)} ${String(n).padStart(4)} ${verdict}`;
|
|
1363
1420
|
}),
|
|
1421
|
+
// The two things the reason alone could not say.
|
|
1422
|
+
//
|
|
1423
|
+
// Both exist because the per-reason table above was the steering wheel
|
|
1424
|
+
// and it pointed at one faculty for a class holding three, and counted
|
|
1425
|
+
// the simulator's own failures towards a perception phase.
|
|
1426
|
+
b.verdicts?.length ? '' : null,
|
|
1427
|
+
b.verdicts?.length
|
|
1428
|
+
? 'verification_failed, split by the verdict that fired'
|
|
1429
|
+
+ (b.derived?.verdicts
|
|
1430
|
+
? ` (${b.derived.verdicts} of ${b.verdicts.reduce((a, v) => a + v.count, 0)} derived from the detail text, not recorded at the time):`
|
|
1431
|
+
: ':')
|
|
1432
|
+
: null,
|
|
1433
|
+
...(b.verdicts ?? []).slice(0, 8).map((v) => {
|
|
1434
|
+
const faculty = metrics.VERDICT_FACULTY[v.name];
|
|
1435
|
+
return ` ${v.name.padEnd(20)} ${String(v.count).padStart(4)} `
|
|
1436
|
+
+ (faculty
|
|
1437
|
+
? `points at: ${faculty}`
|
|
1438
|
+
: 'no faculty follows from this verdict alone — see metrics.VERDICT_FACULTY');
|
|
1439
|
+
}),
|
|
1440
|
+
b.device_total ? '' : null,
|
|
1441
|
+
b.device_total
|
|
1442
|
+
? `${b.device_total} escalation(s) were the DEVICE, not the code — item 173, and not evidence for any faculty`
|
|
1443
|
+
+ `${b.derived?.device ? ` (${b.derived.device} derived from the detail text)` : ''}:`
|
|
1444
|
+
: null,
|
|
1445
|
+
...(b.device ?? []).slice(0, 6).map((d) => ` ${String(d.count).padStart(4)} ${d.name}`),
|
|
1364
1446
|
b.total ? '' : null,
|
|
1365
1447
|
b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
|
|
1366
1448
|
// Said out loud rather than left for someone to discover: the rate is
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
// Is a failure the device's fault or the code's?
|
|
2
|
+
//
|
|
3
|
+
// Moved out of `scripts/` on 2026-09-17 so `src/metrics.js` can use it. The
|
|
4
|
+
// escalation log needed it: 78 of the `verification_failed` records on the
|
|
5
|
+
// bench device are `xcrun simctl openurl` failing, an app that would not
|
|
6
|
+
// launch, or capture stopping — the device, filed under a code faculty and
|
|
7
|
+
// reported as evidence for "sense of time (Phase 11)". A steering wheel that
|
|
8
|
+
// counts item 173's own occurrences as a perception problem points somewhere
|
|
9
|
+
// nobody chose.
|
|
10
|
+
//
|
|
11
|
+
// Deliberately dependency-free. `wedge.js` imports `index.js`, and `metrics.js`
|
|
12
|
+
// is imported *by* `index.js`, so a shared table living in either would close a
|
|
13
|
+
// cycle. This module imports nothing and is imported by both.
|
|
14
|
+
|
|
15
|
+
/** Conditions that are the simulator, not the code. Each seen in a real run. */
|
|
16
|
+
export const DEVICE_STATE = [
|
|
17
|
+
[/NSPOSIXErrorDomain.*code=?\s*60|Operation timed out/i, 'simctl stopped answering (NSPOSIXErrorDomain 60)'],
|
|
18
|
+
[/did not produce a frame|produced no frame in \d+s/i, 'the daemon is up and the display renders nothing'],
|
|
19
|
+
[/Timeout waiting for screen surfaces|display surface is not answering|display surface could not be read/i, 'the display surface is wedged'],
|
|
20
|
+
[/no frames buffered|capture is wedged/i, 'capture stopped'],
|
|
21
|
+
[/the second app never launched|could not be dispatched/i, 'an app would not launch'],
|
|
22
|
+
// A launched app that never comes to the front, seen as the tour waiting for
|
|
23
|
+
// one of its landmarks on a screen that is showing a clock and nothing else.
|
|
24
|
+
//
|
|
25
|
+
// Measured on a runner: `ok launch — launched com.apple.Preferences
|
|
26
|
+
// (relaunched)` followed by `waited 8000ms for General: "General" is not on
|
|
27
|
+
// this screen. Visible: 10:50, .?o (the screen has not moved for 6181ms)`.
|
|
28
|
+
// Two labels, one of them a clock, on a still screen — the device is not
|
|
29
|
+
// presenting the app, and the guard called that a check failing on its
|
|
30
|
+
// merits and declined to revive.
|
|
31
|
+
//
|
|
32
|
+
// Deliberately narrow. It requires the wait to have failed AND the screen to
|
|
33
|
+
// have been still AND almost nothing readable: a tour that genuinely asks for
|
|
34
|
+
// the wrong label has a screen full of other labels, and must keep failing
|
|
35
|
+
// rather than being retried into a pass.
|
|
36
|
+
[
|
|
37
|
+
/never arrived[\s\S]*?Visible:[^\n]{0,24}\(the screen has not moved for \d+ms/i,
|
|
38
|
+
'a launched app never came to the front (the screen shows a clock and nothing else)',
|
|
39
|
+
],
|
|
40
|
+
// The same condition, now said outright by the step that suffered it instead
|
|
41
|
+
// of inferred from the shape of the screen afterwards. Item 169 gave `launch`
|
|
42
|
+
// a pid to compare, so a launch that starts a process and never fronts it
|
|
43
|
+
// reports itself; this signature fires on the cause rather than on a
|
|
44
|
+
// consequence that had to be recognised by "two labels, one a clock".
|
|
45
|
+
//
|
|
46
|
+
// It cannot be triggered by a tour asking for the wrong label — only a failed
|
|
47
|
+
// launch emits this sentence — so it needs none of the narrowing above.
|
|
48
|
+
[
|
|
49
|
+
/never came to the front within \d+ms/i,
|
|
50
|
+
'a launched app never came to the front (the launch said so itself, by pid)',
|
|
51
|
+
],
|
|
52
|
+
// simctl itself stopped answering, and said so in simframe's own words: the
|
|
53
|
+
// process was killed at the timeout rather than refusing the request.
|
|
54
|
+
//
|
|
55
|
+
// Three runs of the 2026-09-17 bench died this way — 95.6 s each — and every
|
|
56
|
+
// one of them was written to `flows.jsonl` with `device_cause: null`, so the
|
|
57
|
+
// number CI gates on counted a simulator that had stopped answering as the
|
|
58
|
+
// code getting things wrong. That is the same fault item 173 recorded as
|
|
59
|
+
// fixed, still open for this signature because nothing here matched it.
|
|
60
|
+
[
|
|
61
|
+
/did not return within \d+s \(killed by simframe/i,
|
|
62
|
+
'simctl stopped answering and had to be killed at the timeout',
|
|
63
|
+
],
|
|
64
|
+
// Seen on the v0.14.3 bench run: `could not launch com.apple.Preferences:
|
|
65
|
+
// The system shell (SpringBoard:36454) probably crashed.` The guest's window
|
|
66
|
+
// server going down is the device, not the check, and nothing here matched it.
|
|
67
|
+
[
|
|
68
|
+
/system shell \(SpringBoard[^)]*\) probably crashed/i,
|
|
69
|
+
"the guest's SpringBoard crashed, so nothing can be fronted",
|
|
70
|
+
],
|
|
71
|
+
];
|
|
72
|
+
|
|
73
|
+
/** The condition this output shows, or null when the check failed on its merits. */
|
|
74
|
+
export function deviceCause(text) {
|
|
75
|
+
return DEVICE_STATE.find(([re]) => re.test(String(text ?? '')))?.[1] ?? null;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* The one device failure that is known to heal by itself.
|
|
80
|
+
*
|
|
81
|
+
* The guest's window server dies, the launch that was in flight fails, and
|
|
82
|
+
* SpringBoard comes back a few seconds later. Item 173 spent a dozen
|
|
83
|
+
* occurrences unable to observe it for exactly that reason — by the time
|
|
84
|
+
* anything looked, `diagnose` said `healthy`, fusion 0.857.
|
|
85
|
+
*
|
|
86
|
+
* Measured on `326464A4` on 2026-09-18, six occurrences in one run: waiting and
|
|
87
|
+
* launching again recovered **6 of 6**, in 5.7-7.9 s (median ~6.3 s). The
|
|
88
|
+
* remedy in use until now was `simframe revive`, a ~40 s device restart — six
|
|
89
|
+
* times the cost, for a fault that was already over.
|
|
90
|
+
*
|
|
91
|
+
* Separate from `deviceCause` on purpose. Every entry there says "this is the
|
|
92
|
+
* device"; this one says "and it will be back". `simctl did not return within
|
|
93
|
+
* 90s` is the device too and is deliberately NOT here: it was never once
|
|
94
|
+
* observed to recover, and a retry costs another 90 s to find that out.
|
|
95
|
+
*/
|
|
96
|
+
export const SHELL_CRASH = /system shell \(SpringBoard[^)]*\) probably crashed/i;
|
|
97
|
+
|
|
98
|
+
/** Did this failure come from the guest shell dying under us? */
|
|
99
|
+
export function shellCrashed(text) {
|
|
100
|
+
return SHELL_CRASH.test(String(text ?? ''));
|
|
101
|
+
}
|
package/src/index.js
CHANGED
|
@@ -1397,22 +1397,76 @@ export async function getFrameAt(deviceQuery, { msAgo = 0, options } = {}) {
|
|
|
1397
1397
|
* Wait, briefly, for a frame that is holding still. Returns whatever the newest
|
|
1398
1398
|
* frame is once the screen settles or the budget runs out, saying which.
|
|
1399
1399
|
*/
|
|
1400
|
-
|
|
1400
|
+
/**
|
|
1401
|
+
* When the stillness a state reports began, or null when that cannot be told.
|
|
1402
|
+
*
|
|
1403
|
+
* `stableForMs` is measured as of the frame's capture, so the quiet period
|
|
1404
|
+
* started `stableForMs` before `capturedAt`. Exported because the rule built on
|
|
1405
|
+
* it below is the interesting part and deserves to be testable without a device.
|
|
1406
|
+
*/
|
|
1407
|
+
export function stillnessBegan(state) {
|
|
1408
|
+
if (!Number.isFinite(state?.capturedAt) || !Number.isFinite(state?.stableForMs)) return null;
|
|
1409
|
+
return state.capturedAt - state.stableForMs;
|
|
1410
|
+
}
|
|
1411
|
+
|
|
1412
|
+
/**
|
|
1413
|
+
* Has this screen been still *since we acted*, or was it still before we did?
|
|
1414
|
+
*
|
|
1415
|
+
* The distinction the settle detector was missing. It answers "how long has the
|
|
1416
|
+
* screen been quiet" honestly and has no idea that the quiet it is describing
|
|
1417
|
+
* belongs to the screen the caller has just left.
|
|
1418
|
+
*
|
|
1419
|
+
* Only a stillness we can **prove** predates the action is rejected. When the
|
|
1420
|
+
* timestamps are missing this returns true, which is the previous behaviour —
|
|
1421
|
+
* this may only ever add refusals it can demonstrate, never turn an unknown
|
|
1422
|
+
* into a wait.
|
|
1423
|
+
*/
|
|
1424
|
+
export function stillSinceActing(state, actedAt) {
|
|
1425
|
+
if (!Number.isFinite(actedAt)) return true;
|
|
1426
|
+
const began = stillnessBegan(state);
|
|
1427
|
+
if (began == null) return true;
|
|
1428
|
+
return began >= actedAt;
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1431
|
+
export async function settledState(udid, { settleMs = MEMORY_SETTLE_MS, timeoutMs = 1500, since } = {}) {
|
|
1401
1432
|
const p = store.paths(udid);
|
|
1402
1433
|
const deadline = Date.now() + timeoutMs;
|
|
1434
|
+
// What we are settling *after*. A caller that knows may say; otherwise it is
|
|
1435
|
+
// the last launch, openUrl or gesture this device was given.
|
|
1436
|
+
const actedAt = since ?? store.lastActionAt(udid);
|
|
1403
1437
|
let state = store.readJson(p.state);
|
|
1438
|
+
let stale = false;
|
|
1404
1439
|
while (Date.now() < deadline) {
|
|
1405
1440
|
state = store.readJson(p.state) ?? state;
|
|
1406
|
-
//
|
|
1407
|
-
//
|
|
1408
|
-
//
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1441
|
+
// **Stillness from before the action is not settlement.** Measured on the
|
|
1442
|
+
// bench device: 283ms after a Settings launch the state said `settled` with
|
|
1443
|
+
// `stableForMs: 4427` over a screen holding **zero** elements, which filled
|
|
1444
|
+
// 2.3 seconds later; 408ms after `tap General` it said `settled` with
|
|
1445
|
+
// `stableForMs: 7753` over the 23 elements of Settings root. Both readings
|
|
1446
|
+
// were honest about the number and wrong about the screen.
|
|
1447
|
+
//
|
|
1448
|
+
// Field-reported independently as the highest-priority class here: a read
|
|
1449
|
+
// that returns chrome with the content missing and no loading marker is
|
|
1450
|
+
// indistinguishable from a screen that is genuinely empty, so an agent
|
|
1451
|
+
// reports "this filter returns zero results" and means it.
|
|
1452
|
+
const fresh = stillSinceActing(state, actedAt);
|
|
1453
|
+
if (!fresh) stale = true;
|
|
1454
|
+
else {
|
|
1455
|
+
// The daemon runs a real settle detector that can tell a spinner from a
|
|
1456
|
+
// still screen. Prefer it; the duration check is the fallback for the
|
|
1457
|
+
// simctl engine, which has no such thing.
|
|
1458
|
+
if (state?.settled === true) return { state, settled: true, waitedForAction: stale };
|
|
1459
|
+
if (state && state.settled === undefined && state.stableForMs >= settleMs) {
|
|
1460
|
+
return { state, settled: true, waitedForAction: stale };
|
|
1461
|
+
}
|
|
1412
1462
|
}
|
|
1413
1463
|
await sleep(40);
|
|
1414
1464
|
}
|
|
1415
|
-
|
|
1465
|
+
// `settled: false` is not a failure and never was — callers use the state and
|
|
1466
|
+
// decline to persist a map built from it. What is new is that this can now be
|
|
1467
|
+
// false because the screen has not been seen to move since the action, which
|
|
1468
|
+
// is a different and more honest reason than "it is still moving".
|
|
1469
|
+
return { state, settled: false, ...(stale ? { stillnessPredatesAction: true } : {}) };
|
|
1416
1470
|
}
|
|
1417
1471
|
|
|
1418
1472
|
/**
|
|
@@ -1459,6 +1513,33 @@ export function offScreenMatch(targets, query, points) {
|
|
|
1459
1513
|
return hit.status === 'ambiguous' && hit.alternatives?.length ? hit.alternatives[0] : null;
|
|
1460
1514
|
}
|
|
1461
1515
|
|
|
1516
|
+
/**
|
|
1517
|
+
* Which way a target is outside the viewport, and by which measure.
|
|
1518
|
+
*
|
|
1519
|
+
* `offViewport` has checked both axes since it was written; the sentence
|
|
1520
|
+
* reporting it only ever named `y` and the screen *height*. So a tab in a
|
|
1521
|
+
* horizontally-scrolling strip was reported as *"it is at y=143 on a 874pt
|
|
1522
|
+
* screen"* — a coordinate plainly inside the screen, next to a conclusion that
|
|
1523
|
+
* it is not in view, with four of its siblings visible. Field-reported, and the
|
|
1524
|
+
* reporter's summary is the right one: the honest half was right and only the
|
|
1525
|
+
* axis was wrong.
|
|
1526
|
+
*
|
|
1527
|
+
* Vertical is checked first because it is overwhelmingly the common case, and
|
|
1528
|
+
* because `scrollTo` reasons vertically — a caller told "below the fold" and a
|
|
1529
|
+
* caller told "past the right edge" do different things next, which is the
|
|
1530
|
+
* whole reason to say which.
|
|
1531
|
+
*/
|
|
1532
|
+
export function offScreenAxis(target, points) {
|
|
1533
|
+
const { width, height } = points ?? {};
|
|
1534
|
+
if (Number.isFinite(height) && (target.y < 0 || target.y > height)) {
|
|
1535
|
+
return { axis: 'y', at: Math.round(target.y), extent: Math.round(height), edge: target.y < 0 ? 'above' : 'below' };
|
|
1536
|
+
}
|
|
1537
|
+
if (Number.isFinite(width) && (target.x < 0 || target.x > width)) {
|
|
1538
|
+
return { axis: 'x', at: Math.round(target.x), extent: Math.round(width), edge: target.x < 0 ? 'left of' : 'right of' };
|
|
1539
|
+
}
|
|
1540
|
+
return null;
|
|
1541
|
+
}
|
|
1542
|
+
|
|
1462
1543
|
/**
|
|
1463
1544
|
* Which sensors a read asks for by default.
|
|
1464
1545
|
*
|
|
@@ -1500,9 +1581,52 @@ export function sensorMode(options) {
|
|
|
1500
1581
|
*/
|
|
1501
1582
|
const nameFor = (t) => t.label || t.identifier || null;
|
|
1502
1583
|
|
|
1584
|
+
/**
|
|
1585
|
+
* Did screen memory alone produce this miss?
|
|
1586
|
+
*
|
|
1587
|
+
* A recall that finds the screen and not the target tags `ambiguous_intent`
|
|
1588
|
+
* without `ambiguous` — "I know this screen, and what you asked for is not on
|
|
1589
|
+
* it". That is the only miss worth re-asking, because it is the only one whose
|
|
1590
|
+
* answer came from a file rather than from the device. A miss off a map that
|
|
1591
|
+
* was just built has already read both sensors, and a genuine ambiguity is not
|
|
1592
|
+
* settled by reading again.
|
|
1593
|
+
*/
|
|
1594
|
+
export function memoryMiss(err) {
|
|
1595
|
+
const why = metrics.escalationOf(err);
|
|
1596
|
+
return Boolean(why && why.reason === 'ambiguous_intent' && !why.ambiguous);
|
|
1597
|
+
}
|
|
1598
|
+
|
|
1503
1599
|
export async function locate(deviceQuery, query, opts = {}) {
|
|
1504
1600
|
if (sensorMode(opts.options) !== 'ax-first' || opts.useOcr === false || opts.escalated) {
|
|
1505
|
-
return locateWith(deviceQuery, query, opts);
|
|
1601
|
+
if (opts.refresh || opts.escalated) return locateWith(deviceQuery, query, opts);
|
|
1602
|
+
try {
|
|
1603
|
+
return await locateWith(deviceQuery, query, opts);
|
|
1604
|
+
} catch (err) {
|
|
1605
|
+
// **Memory may confirm, never deny.** Measured on the bench device, 8 of
|
|
1606
|
+
// 8 and 4 of 4 in two separate runs: at the instant a flow's next step
|
|
1607
|
+
// asks, a recall says "not on this screen" in 25 ms and a fresh read
|
|
1608
|
+
// finds the target 1.8 s later, on the same device, without anything
|
|
1609
|
+
// touching it in between.
|
|
1610
|
+
//
|
|
1611
|
+
// Why the recall is wrong, and it is not staleness in the usual sense:
|
|
1612
|
+
// the frame it keys on was captured 47-124 ms *after* the tap — the first
|
|
1613
|
+
// frame of the push animation, which still looks like the screen being
|
|
1614
|
+
// left. Capture is damage-driven, so that frame then goes still,
|
|
1615
|
+
// `settledState` calls it settled at 500-700 ms of stillness, and
|
|
1616
|
+
// `recallNearest` — deliberately tolerant, because a list with new rows is
|
|
1617
|
+
// still the same screen — matches it back to the previous screen and
|
|
1618
|
+
// answers out of that screen's stored element list. The screen itself
|
|
1619
|
+
// arrives about 750 ms later.
|
|
1620
|
+
//
|
|
1621
|
+
// The cost is paid only on a miss, which today aborts the batch and buys
|
|
1622
|
+
// a ~20 s model round trip. 1.8 s to be sure is the cheaper mistake. The
|
|
1623
|
+
// hit path — the one the speed argument rests on — is untouched.
|
|
1624
|
+
//
|
|
1625
|
+
// This recovery already existed for `ax-first` (below) and had never run
|
|
1626
|
+
// in the default sensor mode, which is `full`.
|
|
1627
|
+
if (!memoryMiss(err)) throw err;
|
|
1628
|
+
return locateWith(deviceQuery, query, { ...opts, refresh: true, escalated: true });
|
|
1629
|
+
}
|
|
1506
1630
|
}
|
|
1507
1631
|
const options = opts;
|
|
1508
1632
|
try {
|
|
@@ -1723,8 +1847,12 @@ async function locateWith(
|
|
|
1723
1847
|
if (offScreen) {
|
|
1724
1848
|
throw metrics.tag(
|
|
1725
1849
|
new Error(
|
|
1726
|
-
`"${query}" is in the tree but not in view —
|
|
1727
|
-
|
|
1850
|
+
`"${query}" is in the tree but not in view — ${(() => {
|
|
1851
|
+
const off = offScreenAxis(offScreen, points);
|
|
1852
|
+
if (!off) return `it is at ${Math.round(offScreen.x)},${Math.round(offScreen.y)}`;
|
|
1853
|
+
return `it is ${off.edge} the viewport, at ${off.axis}=${off.at}`
|
|
1854
|
+
+ ` on a ${off.extent}pt ${off.axis === 'y' ? 'tall' : 'wide'} screen`;
|
|
1855
|
+
})()}. Scroll to it (sim_scroll_to) rather than waiting;`
|
|
1728
1856
|
+ ' waiting cannot bring it into view.',
|
|
1729
1857
|
),
|
|
1730
1858
|
from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',
|