simframe 0.7.2 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -4
- package/flows/hpi-suite.json +68 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +49 -1
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +7 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +11 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +48 -0
- package/native/simframed/Sources/simframed/main.swift +128 -73
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +40 -0
- package/package.json +2 -1
- package/scripts/bench-hpi.mjs +254 -0
- package/scripts/check-package.mjs +7 -0
- package/scripts/ci-memory.mjs +15 -2
- package/src/actions.js +184 -4
- package/src/baseline.js +333 -0
- package/src/cli.js +336 -2
- package/src/daemon.js +9 -0
- package/src/fingerprint.js +7 -1
- package/src/graph.js +162 -1
- package/src/index.js +118 -20
- package/src/input.js +111 -1
- package/src/intent.js +11 -2
- package/src/matching.js +81 -2
- package/src/mcp.js +14 -1
- package/src/metrics.js +499 -0
- package/src/navigate.js +44 -7
- package/src/platform/android.js +27 -0
- package/src/platform/index.js +3 -0
- package/src/platform/ios.js +63 -0
- package/src/screenmap.js +36 -14
- package/src/view.js +4 -3
package/src/cli.js
CHANGED
|
@@ -1,12 +1,16 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import fs from 'node:fs';
|
|
3
|
+
import os from 'node:os';
|
|
3
4
|
import path from 'node:path';
|
|
4
5
|
import { runDaemon, DEFAULTS } from './daemon.js';
|
|
5
|
-
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, toolchainChecks } from './platform/index.js';
|
|
6
|
+
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
|
|
6
7
|
import * as actions from './actions.js';
|
|
7
8
|
import * as api from './index.js';
|
|
8
9
|
import * as input from './input.js';
|
|
10
|
+
import * as baseline from './baseline.js';
|
|
11
|
+
import * as metrics from './metrics.js';
|
|
9
12
|
import * as navigate from './navigate.js';
|
|
13
|
+
import { decodePng } from './png.js';
|
|
10
14
|
import * as store from './store.js';
|
|
11
15
|
import * as view from './view.js';
|
|
12
16
|
|
|
@@ -36,6 +40,11 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
36
40
|
simframe type <text> enter text (exact; uses the pasteboard)
|
|
37
41
|
simframe keys <text> send key events instead (layout-dependent)
|
|
38
42
|
simframe press <button> a hardware button, e.g. home
|
|
43
|
+
simframe baseline record <flow> record a human performing a flow (see below)
|
|
44
|
+
simframe baseline summarize <flow> write the median/IQR baseline for it
|
|
45
|
+
simframe baseline list recorded runs per flow, and what is committed
|
|
46
|
+
simframe hpi [device] Human Parity Index, per flow and overall
|
|
47
|
+
simframe escalations [device] why simframe handed decisions back, by reason
|
|
39
48
|
simframe devices list simulators
|
|
40
49
|
simframe doctor check that this machine can capture
|
|
41
50
|
(--strict, or SIMFRAME_STRICT=1, makes any
|
|
@@ -46,6 +55,21 @@ Selectors — anywhere a control is named
|
|
|
46
55
|
"Save" a label or a phrase, resolved by intent (verbs, typos, synonyms)
|
|
47
56
|
@120,400 raw point coordinates
|
|
48
57
|
|
|
58
|
+
Measuring against a human — the Human Parity Index
|
|
59
|
+
|
|
60
|
+
A flow's agent time is measured every time it runs; the human half has to be
|
|
61
|
+
recorded once, by a person, on the same simulator:
|
|
62
|
+
|
|
63
|
+
simframe baseline record settings-larger-text --device=<udid> --runs=5
|
|
64
|
+
simframe baseline summarize settings-larger-text
|
|
65
|
+
simframe hpi --device=<udid>
|
|
66
|
+
|
|
67
|
+
\`record\` puts the device on the home screen, waits for you to start, and
|
|
68
|
+
waits again for you to stop. Wall time is measured between those two; the
|
|
69
|
+
step count is derived from screen transitions, because a human tapping the
|
|
70
|
+
Simulator window leaves no HID log to read. Five runs is the recommendation
|
|
71
|
+
and three is the floor. The flows live in flows/hpi-suite.json.
|
|
72
|
+
|
|
49
73
|
Options
|
|
50
74
|
--device=<udid|name> simulator to target (default: the booted one)
|
|
51
75
|
--json machine-readable output — on every command
|
|
@@ -96,6 +120,7 @@ The reliable pattern around an action is:
|
|
|
96
120
|
const VALUE_FLAGS = new Set([
|
|
97
121
|
'ago', 'count', 'detail', 'device', 'durationMs', 'engine', 'filter', 'fps', 'index', 'maxDim',
|
|
98
122
|
'mode', 'out', 'ringSize', 'since', 'spanMs', 'stableMs', 'timeoutMs',
|
|
123
|
+
'last', 'runs', 'suite', 'keepLast', 'reason',
|
|
99
124
|
]);
|
|
100
125
|
|
|
101
126
|
function parseArgs(argv) {
|
|
@@ -154,6 +179,40 @@ async function mapText(device, options, identity) {
|
|
|
154
179
|
}
|
|
155
180
|
}
|
|
156
181
|
|
|
182
|
+
/**
|
|
183
|
+
* Wait for Enter, on a terminal or on a pipe.
|
|
184
|
+
*
|
|
185
|
+
* `readline/promises` looked like the obvious choice and threw "readline was
|
|
186
|
+
* closed" the first time it was asked a question after piped input ran out —
|
|
187
|
+
* which is how a smoke test of `baseline record` would have handed a stack
|
|
188
|
+
* trace to the person recording. Buffered lines are queued, and a closed
|
|
189
|
+
* stream is reported as what it is rather than thrown from inside a library.
|
|
190
|
+
*/
|
|
191
|
+
async function lineReader() {
|
|
192
|
+
const readline = await import('node:readline');
|
|
193
|
+
const rl = readline.createInterface({ input: process.stdin, terminal: Boolean(process.stdin.isTTY) });
|
|
194
|
+
const queue = [];
|
|
195
|
+
const waiters = [];
|
|
196
|
+
let closed = false;
|
|
197
|
+
rl.on('line', (line) => (waiters.length ? waiters.shift()({ line }) : queue.push(line)));
|
|
198
|
+
rl.on('close', () => {
|
|
199
|
+
closed = true;
|
|
200
|
+
while (waiters.length) waiters.shift()({ closed: true });
|
|
201
|
+
});
|
|
202
|
+
return {
|
|
203
|
+
async enter(prompt) {
|
|
204
|
+
process.stdout.write(prompt);
|
|
205
|
+
const got = queue.length ? { line: queue.shift() } : closed ? { closed: true } : await new Promise((r) => waiters.push(r));
|
|
206
|
+
if (got.closed) {
|
|
207
|
+
throw new Error('stdin closed before the run ended — baseline record needs an interactive terminal');
|
|
208
|
+
}
|
|
209
|
+
process.stdout.write('\n');
|
|
210
|
+
return got.line;
|
|
211
|
+
},
|
|
212
|
+
close: () => rl.close(),
|
|
213
|
+
};
|
|
214
|
+
}
|
|
215
|
+
|
|
157
216
|
/** A step result, the same shape in every command that runs steps. */
|
|
158
217
|
const stepLine = (r) => {
|
|
159
218
|
const settle = r.settled ? (r.settled.ok ? ` (settled ${r.settled.waitedMs}ms)` : ' (never settled)') : '';
|
|
@@ -243,6 +302,11 @@ async function main() {
|
|
|
243
302
|
`stopped ${stopped} daemon${stopped === 1 ? '' : 's'}` +
|
|
244
303
|
(inUse ? `; left ${inUse} in use by another client (pass --force to stop anyway)` : ''),
|
|
245
304
|
);
|
|
305
|
+
// Asked to stop one device, refused, and exited 0 — which is a success
|
|
306
|
+
// code for work not done, and a script checking `$?` could not tell the
|
|
307
|
+
// difference. `--all` is informational by nature, so it keeps exiting 0
|
|
308
|
+
// when it skips a device somebody else holds.
|
|
309
|
+
if (!flags.all && inUse && !stopped) process.exitCode = 1;
|
|
246
310
|
return;
|
|
247
311
|
}
|
|
248
312
|
|
|
@@ -300,13 +364,23 @@ async function main() {
|
|
|
300
364
|
}
|
|
301
365
|
|
|
302
366
|
case 'state': {
|
|
303
|
-
const res = await api.getState(device, { since: flags.since, options });
|
|
367
|
+
const res = await api.getState(device, { since: flags.since, options, inputHealth: true });
|
|
304
368
|
if (flags.json) {
|
|
305
369
|
console.log(JSON.stringify({ ...res.state, history: undefined, ageMs: res.ageMs, since: res.since, live: res.live }, null, 2));
|
|
306
370
|
} else {
|
|
307
371
|
const s = res.state;
|
|
308
372
|
const out = [];
|
|
309
373
|
if (!res.live.ok) out.push(`WARNING: ${res.live.note}`);
|
|
374
|
+
// A cause, rather than five silent no-ops. Every tap on a stale
|
|
375
|
+
// session is dispatched successfully and moves nothing.
|
|
376
|
+
if (res.input?.stale) out.push(`input: stale — ${res.input.reason}`);
|
|
377
|
+
if (res.timing?.samples) {
|
|
378
|
+
out.push(
|
|
379
|
+
`timing: this screen usually arrives in ${res.timing.edge_p50}ms (p95 ${res.timing.edge_p95}ms, `
|
|
380
|
+
+ `${res.timing.samples} samples); ${res.timing.elapsed_ms}ms since the last change`
|
|
381
|
+
+ (res.timing.slower_than_usual ? ` — ${res.timing.note}` : ''),
|
|
382
|
+
);
|
|
383
|
+
}
|
|
310
384
|
out.push(`${res.device.name} frame #${s.seq} age ${res.ageMs}ms ${s.width}x${s.height}`);
|
|
311
385
|
out.push(`hash ${s.hash} stable ${s.stableForMs}ms`);
|
|
312
386
|
if (res.since?.kind === 'history') {
|
|
@@ -719,6 +793,188 @@ async function main() {
|
|
|
719
793
|
return;
|
|
720
794
|
}
|
|
721
795
|
|
|
796
|
+
case 'baseline': {
|
|
797
|
+
const [sub, name] = positional;
|
|
798
|
+
const suite = baseline.loadSuite(flags.suite ?? baseline.SUITE_FILE);
|
|
799
|
+
|
|
800
|
+
if (sub === 'list' || sub == null) {
|
|
801
|
+
const dev = await resolveDevice(flags.device);
|
|
802
|
+
const committed = baseline.readBaselines();
|
|
803
|
+
const rows = suite.map((f) => {
|
|
804
|
+
const runs = baseline.readRuns(dev.udid, f.name);
|
|
805
|
+
return {
|
|
806
|
+
flow: f.name,
|
|
807
|
+
recorded_runs: runs.length,
|
|
808
|
+
committed: Boolean(committed[f.name]),
|
|
809
|
+
human_median_ms: committed[f.name]?.wall_time_ms?.p50 ?? null,
|
|
810
|
+
min_steps: f.minSteps ?? null,
|
|
811
|
+
};
|
|
812
|
+
});
|
|
813
|
+
emit(flags, rows, rows.map((r) =>
|
|
814
|
+
`${r.flow.padEnd(24)} ${String(r.recorded_runs).padStart(2)} run${r.recorded_runs === 1 ? ' ' : 's'}` +
|
|
815
|
+
` ${r.committed ? `committed, human p50 ${r.human_median_ms}ms` : 'not committed'}`));
|
|
816
|
+
return;
|
|
817
|
+
}
|
|
818
|
+
|
|
819
|
+
if (sub === 'exclude') {
|
|
820
|
+
if (!name) throw new Error('usage: simframe baseline exclude <flow> --keep-last=<n> [--reason="..."]');
|
|
821
|
+
const dev = await resolveDevice(flags.device);
|
|
822
|
+
baseline.flowFrom(suite, name);
|
|
823
|
+
const keepLast = num(flags.keepLast, NaN);
|
|
824
|
+
if (!Number.isFinite(keepLast)) throw new Error('pass --keep-last=<n>: how many of the most recent runs to keep');
|
|
825
|
+
const res = baseline.excludeRuns(dev.udid, name, { keepLast, reason: flags.reason ? String(flags.reason) : undefined });
|
|
826
|
+
emit(flags, res, `${name}: ${res.excluded} of ${res.total} run(s) marked excluded — the runs stay in the log, carrying why`);
|
|
827
|
+
return;
|
|
828
|
+
}
|
|
829
|
+
|
|
830
|
+
if (sub === 'summarize') {
|
|
831
|
+
if (!name) throw new Error('usage: simframe baseline summarize <flow>');
|
|
832
|
+
const dev = await resolveDevice(flags.device);
|
|
833
|
+
const flow = baseline.flowFrom(suite, name);
|
|
834
|
+
const runs = baseline.readRuns(dev.udid, name);
|
|
835
|
+
const res = baseline.summarizeRuns(name, runs, { minSteps: flow.minSteps ?? null, device: dev.udid });
|
|
836
|
+
if (!res.ok) {
|
|
837
|
+
emit(flags, res, `${runs.length} recorded run${runs.length === 1 ? '' : 's'} for "${name}" — ` +
|
|
838
|
+
`${baseline.MIN_RUNS} is the floor and ${baseline.WANT_RUNS} is the recommendation. ` +
|
|
839
|
+
`Record more: simframe baseline record ${name} --device=${dev.udid}`);
|
|
840
|
+
process.exitCode = 1;
|
|
841
|
+
return;
|
|
842
|
+
}
|
|
843
|
+
const file = flags.out ? baseline.writeSummary(res.summary, { dir: path.dirname(flags.out) }) : baseline.writeSummary(res.summary);
|
|
844
|
+
const w = res.summary.wall_time_ms;
|
|
845
|
+
emit(flags, { ...res.summary, file }, [
|
|
846
|
+
`${name}: ${res.summary.runs} runs`,
|
|
847
|
+
` wall time p50 ${w.p50}ms IQR ${w.p25}–${w.p75}ms range ${w.min}–${w.max}ms`,
|
|
848
|
+
` steps p50 ${res.summary.steps_observed.p50} (from screen transitions; min_steps ${res.summary.min_steps ?? '?'} from the flow)`,
|
|
849
|
+
res.summary.runs_with_incomplete_history
|
|
850
|
+
? ` WARNING ${res.summary.runs_with_incomplete_history} run(s) outran the 90s frame history; their step counts are undercounts`
|
|
851
|
+
: null,
|
|
852
|
+
` wrote ${file}`,
|
|
853
|
+
]);
|
|
854
|
+
return;
|
|
855
|
+
}
|
|
856
|
+
|
|
857
|
+
if (sub === 'record') {
|
|
858
|
+
if (!name) throw new Error('usage: simframe baseline record <flow>');
|
|
859
|
+
const flow = baseline.flowFrom(suite, name);
|
|
860
|
+
const dev = await resolveDevice(flags.device);
|
|
861
|
+
const wanted = Math.max(1, num(flags.runs, 1));
|
|
862
|
+
const rl = await lineReader();
|
|
863
|
+
const done = [];
|
|
864
|
+
try {
|
|
865
|
+
console.log(`${name} on ${dev.name} (${dev.udid})`);
|
|
866
|
+
console.log(`${flow.note ?? ''}\n`);
|
|
867
|
+
console.log('Do this, at your natural pace:');
|
|
868
|
+
for (const [i, line] of (flow.human ?? []).entries()) console.log(` ${i + 1}. ${line}`);
|
|
869
|
+
console.log(`\nThe shortest route is ${flow.minSteps} steps. Practise once or twice first —`);
|
|
870
|
+
console.log('a baseline should measure a tester who knows the flow, not one discovering it.\n');
|
|
871
|
+
for (let run = 1; run <= wanted; run += 1) {
|
|
872
|
+
const reset = await baseline.resetFor(dev.udid, flow);
|
|
873
|
+
if (reset.failures.length) console.log(` (reset: ${reset.failures.join('; ')})`);
|
|
874
|
+
await rl.enter(`run ${run}/${wanted} — device is on the home screen. Press Enter, then do the flow: `);
|
|
875
|
+
const res = await baseline.recordHumanRun(dev.udid, name, {
|
|
876
|
+
suite,
|
|
877
|
+
options,
|
|
878
|
+
waitForStop: () => rl.enter(` timing... press Enter the moment you are on "${flow.endsOn ?? 'the last screen'}": `),
|
|
879
|
+
});
|
|
880
|
+
done.push(res.run);
|
|
881
|
+
const r = res.run;
|
|
882
|
+
console.log(` ${r.wall_time_ms}ms, ${r.steps_observed} screen transitions` +
|
|
883
|
+
(r.history_complete === false ? ' — WARNING: longer than the frame history, steps undercounted' : ''));
|
|
884
|
+
}
|
|
885
|
+
} finally {
|
|
886
|
+
rl.close();
|
|
887
|
+
}
|
|
888
|
+
const total = baseline.readRuns(dev.udid, name).length;
|
|
889
|
+
emit(flags, { flow: name, recorded: done, runs_on_file: total }, [
|
|
890
|
+
'',
|
|
891
|
+
`${done.length} run${done.length === 1 ? '' : 's'} recorded — ${total} on file for "${name}"`,
|
|
892
|
+
total < baseline.WANT_RUNS
|
|
893
|
+
? `${baseline.WANT_RUNS - total} more would meet the recommendation; ${Math.max(0, baseline.MIN_RUNS - total)} more is the floor`
|
|
894
|
+
: `enough to summarize: simframe baseline summarize ${name} --device=${dev.udid}`,
|
|
895
|
+
]);
|
|
896
|
+
return;
|
|
897
|
+
}
|
|
898
|
+
|
|
899
|
+
throw new Error('usage: simframe baseline <record|summarize|list> [flow]');
|
|
900
|
+
}
|
|
901
|
+
|
|
902
|
+
case 'hpi': {
|
|
903
|
+
const dev = await resolveDevice(flags.device);
|
|
904
|
+
const suite = baseline.loadSuite(flags.suite ?? baseline.SUITE_FILE);
|
|
905
|
+
const names = new Set(suite.map((f) => f.name));
|
|
906
|
+
const all = metrics.readFlows(dev.udid);
|
|
907
|
+
// Named runs only, and only flows this suite defines. An ad-hoc `sim_do`
|
|
908
|
+
// is timed and logged, but it has no human counterpart and averaging it
|
|
909
|
+
// into a parity index would be inventing a comparison.
|
|
910
|
+
let runs = all.filter((f) => f.flow_name && names.has(f.flow_name) && (!flags.flow || f.flow_name === flags.flow));
|
|
911
|
+
// `--last=n` keeps the n most recent runs of each flow. The log is
|
|
912
|
+
// append-only on purpose — it is the trend — but a local log carries
|
|
913
|
+
// runs from a wedged device and from a bug since fixed, and averaging
|
|
914
|
+
// those into today's number describes neither day. CI computes its
|
|
915
|
+
// number from one process's own runs and never needs this.
|
|
916
|
+
if (flags.last) {
|
|
917
|
+
const keep = num(flags.last);
|
|
918
|
+
const perFlow = new Map();
|
|
919
|
+
for (const r of runs) perFlow.set(r.flow_name, [...(perFlow.get(r.flow_name) ?? []), r]);
|
|
920
|
+
runs = [...perFlow.values()].flatMap((rs) => rs.slice(-keep));
|
|
921
|
+
}
|
|
922
|
+
const report = metrics.hpi({ flows: runs, baselines: baseline.readBaselines() });
|
|
923
|
+
if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify(report, null, 2)}\n`);
|
|
924
|
+
if (!runs.length) {
|
|
925
|
+
emit(flags, report, [
|
|
926
|
+
`no runs of any suite flow on this device yet (${all.length} unnamed run${all.length === 1 ? '' : 's'} in the log)`,
|
|
927
|
+
'run the agent side: node scripts/bench-hpi.mjs --device=' + dev.udid,
|
|
928
|
+
]);
|
|
929
|
+
return;
|
|
930
|
+
}
|
|
931
|
+
emit(flags, report, [
|
|
932
|
+
flags.last ? `the last ${num(flags.last)} run(s) of each flow, of ${all.filter((f) => f.flow_name).length} named runs in the log` : null,
|
|
933
|
+
'flow runs agent p50 human p50 HPI_time step_ratio turns esc',
|
|
934
|
+
...report.flows.map((f) =>
|
|
935
|
+
`${f.flow.padEnd(24)} ${String(f.runs).padStart(5)} ${`${f.agent_ms.p50}ms`.padStart(9)} ` +
|
|
936
|
+
`${(f.human_median_ms ? `${f.human_median_ms}ms` : '—').padStart(9)} ` +
|
|
937
|
+
`${(f.hpi_time ?? '—').toString().padStart(8)} ${(f.step_ratio ?? '—').toString().padStart(10)} ` +
|
|
938
|
+
`${(f.model_turns ?? '—').toString().padStart(5)} ${String(f.escalations).padStart(3)}`),
|
|
939
|
+
'',
|
|
940
|
+
`HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs} runs, ` +
|
|
941
|
+
`${report.overall.runs - runs.filter((r) => r.completed && !r.wrong_action_taken).length} not clean)`,
|
|
942
|
+
report.overall.hpi_time == null
|
|
943
|
+
? `HPI_time and HPI need a human baseline — none of ${report.overall.flows_measured} measured flow(s) has one yet.`
|
|
944
|
+
: `HPI_time ${report.overall.hpi_time} (harmonic mean over ${report.overall.flows_with_human_baseline} flow(s)), HPI ${report.overall.hpi}`,
|
|
945
|
+
`step_ratio ${report.overall.step_ratio ?? '—'} (target ≤1.5), model turns per flow ${report.overall.model_turns_median ?? '—'}`,
|
|
946
|
+
flags.out ? `wrote ${flags.out}` : null,
|
|
947
|
+
]);
|
|
948
|
+
return;
|
|
949
|
+
}
|
|
950
|
+
|
|
951
|
+
case 'escalations': {
|
|
952
|
+
const dev = await resolveDevice(flags.device);
|
|
953
|
+
const records = metrics.readEscalations(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
|
|
954
|
+
const b = metrics.breakdown(records);
|
|
955
|
+
if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify(b, null, 2)}\n`);
|
|
956
|
+
emit(flags, b, [
|
|
957
|
+
`${b.total} escalation${b.total === 1 ? '' : 's'} on ${dev.name}`,
|
|
958
|
+
...metrics.REASONS
|
|
959
|
+
.filter((r) => b.by_reason[r])
|
|
960
|
+
.sort((a, c) => b.by_reason[c] - b.by_reason[a])
|
|
961
|
+
.map((r) => ` ${r.padEnd(20)} ${String(b.by_reason[r]).padStart(4)} would be removed by: ${metrics.FACULTY[r]}`),
|
|
962
|
+
b.total ? '' : null,
|
|
963
|
+
b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
|
|
964
|
+
// Said out loud rather than left for someone to discover: the rate is
|
|
965
|
+
// 1.0 while no faculty exists, so the breakdown above is the number
|
|
966
|
+
// that decides the next phase.
|
|
967
|
+
b.total && b.avoidable_escalation_rate === 1
|
|
968
|
+
? ' every reason maps to a faculty that is not built yet, so this rate is 1.0 by construction. The per-reason counts are the steering wheel.'
|
|
969
|
+
: null,
|
|
970
|
+
b.total ? `model turns spent on escalations: ${b.model_turns_spent}` : null,
|
|
971
|
+
b.top_screens.length ? 'top screens:' : null,
|
|
972
|
+
...b.top_screens.map((s) => ` ${s.fingerprint.slice(0, 16).padEnd(18)} ${s.count}`),
|
|
973
|
+
metrics.writeError() ? `WARNING a log write failed: ${metrics.writeError()}` : null,
|
|
974
|
+
]);
|
|
975
|
+
return;
|
|
976
|
+
}
|
|
977
|
+
|
|
722
978
|
case 'devices': {
|
|
723
979
|
const all = await listDevices();
|
|
724
980
|
const shown = flags.all ? all : all.filter((d) => d.state === 'Booted');
|
|
@@ -767,6 +1023,52 @@ async function main() {
|
|
|
767
1023
|
* removed, and a fresh machine without it has not degraded from anything.
|
|
768
1024
|
* Strict fails on warn and fail, never on optional.
|
|
769
1025
|
*/
|
|
1026
|
+
/**
|
|
1027
|
+
* Is the device's display black, or is it only simframe that cannot read it?
|
|
1028
|
+
*
|
|
1029
|
+
* Two very different faults with one symptom, and telling them apart by hand
|
|
1030
|
+
* took an hour: `simctl io screenshot` on a wedged device wrote a valid PNG
|
|
1031
|
+
* whose 3.16 million pixels were all black, in 16.2 s. So the display pipeline
|
|
1032
|
+
* had failed and simframe's read was an accurate report of it.
|
|
1033
|
+
*
|
|
1034
|
+
* Slow on purpose-built-in: this runs once, only when capture has already
|
|
1035
|
+
* declared itself stalled, and 16 s of certainty beats an hour of guessing.
|
|
1036
|
+
*/
|
|
1037
|
+
async function blackScreenProbe(udid) {
|
|
1038
|
+
const file = path.join(os.tmpdir(), `simframe-probe-${Date.now()}.png`);
|
|
1039
|
+
const startedAt = Date.now();
|
|
1040
|
+
try {
|
|
1041
|
+
await screenshot(udid, file);
|
|
1042
|
+
const png = decodePng(fs.readFileSync(file));
|
|
1043
|
+
let lit = 0;
|
|
1044
|
+
for (let i = 0; i < png.data.length; i += 4) {
|
|
1045
|
+
if (png.data[i] > 12 || png.data[i + 1] > 12 || png.data[i + 2] > 12) lit += 1;
|
|
1046
|
+
}
|
|
1047
|
+
const ms = Date.now() - startedAt;
|
|
1048
|
+
const pixels = png.data.length / 4;
|
|
1049
|
+
if (lit === 0) {
|
|
1050
|
+
return {
|
|
1051
|
+
value: 'black',
|
|
1052
|
+
detail: `the device's display is rendering black — every one of ${pixels.toLocaleString()} pixels, `
|
|
1053
|
+
+ `confirmed through Apple's own screenshot path in ${ms}ms. This is the simulator, not simframe: `
|
|
1054
|
+
+ 'it often recovers on its own, and restarting the device also cures it. Re-resolving the '
|
|
1055
|
+
+ 'display port and rebinding the device have both been tried — 223 and 6 times — and neither '
|
|
1056
|
+
+ 'makes any difference, because there is nothing wrong with the handle.',
|
|
1057
|
+
};
|
|
1058
|
+
}
|
|
1059
|
+
return {
|
|
1060
|
+
value: 'readable-by-simctl',
|
|
1061
|
+
detail: `simctl can see ${lit.toLocaleString()} lit pixels of ${pixels.toLocaleString()} in ${ms}ms `
|
|
1062
|
+
+ 'while the daemon cannot read the surface at all. That is a simframe bug, not a wedged simulator — '
|
|
1063
|
+
+ 'worth reporting with this line.',
|
|
1064
|
+
};
|
|
1065
|
+
} catch (err) {
|
|
1066
|
+
return { value: 'unreadable', detail: `even simctl could not screenshot this device: ${err.message}` };
|
|
1067
|
+
} finally {
|
|
1068
|
+
fs.rmSync(file, { force: true });
|
|
1069
|
+
}
|
|
1070
|
+
}
|
|
1071
|
+
|
|
770
1072
|
async function doctor({ json = false, strict = false, device } = {}) {
|
|
771
1073
|
const checks = [];
|
|
772
1074
|
// `level` is 'ok' | 'warn' | 'fail'. A warn means it works but not the way it
|
|
@@ -900,6 +1202,38 @@ async function doctor({ json = false, strict = false, device } = {}) {
|
|
|
900
1202
|
driver.available ? `${driver.name}: ${driver.version}` : driver.reason,
|
|
901
1203
|
{ key: 'input.driver', value: driver.available ? driver.name : null });
|
|
902
1204
|
}
|
|
1205
|
+
// When capture is wedged, say whose fault it is.
|
|
1206
|
+
//
|
|
1207
|
+
// Established the hard way: on a wedged device, Apple's own
|
|
1208
|
+
// `simctl io screenshot` still succeeds — and returns an image with zero
|
|
1209
|
+
// non-black pixels, in 16 seconds instead of one. The simulator's
|
|
1210
|
+
// display pipeline is rendering black; simframe's IOSurface read is not
|
|
1211
|
+
// the thing that broke. Re-resolving the port does not help, and neither
|
|
1212
|
+
// does rebinding the device: both were tried, the second six times.
|
|
1213
|
+
//
|
|
1214
|
+
// That distinction is the whole value of this check. "simframe cannot
|
|
1215
|
+
// read the display" invites someone to debug simframe; "the device's
|
|
1216
|
+
// display is black and Apple's screenshot agrees" tells them to restart
|
|
1217
|
+
// the device. The probe costs one screenshot and only runs when capture
|
|
1218
|
+
// has already given up.
|
|
1219
|
+
const wedged = store.captureHealth(d.udid)?.stalled;
|
|
1220
|
+
if (wedged) {
|
|
1221
|
+
const probe = await blackScreenProbe(d.udid);
|
|
1222
|
+
add(`display (${d.name})`, 'fail', probe.detail, { key: 'display.probe', value: probe.value });
|
|
1223
|
+
}
|
|
1224
|
+
|
|
1225
|
+
// Reported next to the driver it is about. A driver that is present and
|
|
1226
|
+
// working is still useless if it holds a session for a device session
|
|
1227
|
+
// that no longer exists, and that state was invisible: taps were
|
|
1228
|
+
// dispatched successfully and moved nothing, five runs in a row.
|
|
1229
|
+
const session = await input.sessionHealth(d.udid);
|
|
1230
|
+
if (session.stale) {
|
|
1231
|
+
add(`input session (${d.name})`, 'warn', `stale — ${session.reason}. The next action rebuilds it automatically; simframe stop && simframe start does it now`,
|
|
1232
|
+
{ key: 'input.session', value: 'stale' });
|
|
1233
|
+
} else if (caps.input.supported) {
|
|
1234
|
+
add(`input session (${d.name})`, 'ok', session.reason ?? 'current with this device session',
|
|
1235
|
+
{ key: 'input.session', value: 'current' });
|
|
1236
|
+
}
|
|
903
1237
|
add(`text recognition (${d.name})`, 'ok',
|
|
904
1238
|
daemon ? 'simframed (in-process, off the framebuffer)' : 'sips + helper binary');
|
|
905
1239
|
if (!caps.ax.supported) {
|
package/src/daemon.js
CHANGED
|
@@ -92,6 +92,15 @@ export async function runDaemon(device, options = {}) {
|
|
|
92
92
|
let prevSignature = null;
|
|
93
93
|
/** Recent frames, so a caller can diff against whatever it last saw rather
|
|
94
94
|
* than only against the frame that happened to precede this one. */
|
|
95
|
+
// A new capture session inherits no stall.
|
|
96
|
+
//
|
|
97
|
+
// capture-health.json is cleared when a stalled loop captures a frame again,
|
|
98
|
+
// and a loop that dies while stalled never gets to. So a fresh daemon ran
|
|
99
|
+
// healthily while `doctor` reported "stalled for 221s, 60 re-attaches" from
|
|
100
|
+
// its predecessor, and the display probe — correctly, on that input — called
|
|
101
|
+
// it a simframe bug. It was: this one.
|
|
102
|
+
store.writeCaptureHealth(udid, null);
|
|
103
|
+
|
|
95
104
|
let history = [];
|
|
96
105
|
/** seq + timestamp for every frame still on disk, so retention can be thinned by age. */
|
|
97
106
|
let ringIndex = [];
|
package/src/fingerprint.js
CHANGED
|
@@ -24,8 +24,14 @@ import * as regions from './regions.js';
|
|
|
24
24
|
* browser's address bar put "== example.com" into a screen's identity, so
|
|
25
25
|
* a different page read as a different screen, and OCR's ":" and "+" read
|
|
26
26
|
* off icons were identities of their own.
|
|
27
|
+
* 4 — not a token-rule change at all, and bumped anyway: the screen map now
|
|
28
|
+
* merges an OCR reading that sits inside a labelled ax element into that
|
|
29
|
+
* element, so the target list these rules run over is shorter on every
|
|
30
|
+
* list screen. Same rules, different input, therefore different hashes —
|
|
31
|
+
* and a stored hash that can never match again is the quietest kind of
|
|
32
|
+
* wrong, which is what this counter exists to prevent.
|
|
27
33
|
*/
|
|
28
|
-
export const TOKEN_RULES_VERSION =
|
|
34
|
+
export const TOKEN_RULES_VERSION = 4;
|
|
29
35
|
|
|
30
36
|
/** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
|
|
31
37
|
export const GRID = 24;
|
package/src/graph.js
CHANGED
|
@@ -9,11 +9,75 @@ import path from 'node:path';
|
|
|
9
9
|
import { hashDistance } from './analyze.js';
|
|
10
10
|
import { informative } from './refs.js';
|
|
11
11
|
import * as fingerprint from './fingerprint.js';
|
|
12
|
+
import * as metrics from './metrics.js';
|
|
12
13
|
import * as matching from './matching.js';
|
|
13
14
|
import * as store from './store.js';
|
|
14
15
|
|
|
15
16
|
const GRAPH_VERSION = 3;
|
|
16
17
|
|
|
18
|
+
/**
|
|
19
|
+
* How many observed settle durations an edge remembers. Research §7.
|
|
20
|
+
*
|
|
21
|
+
* Fifty is a window, not a history: an app that got faster after an update
|
|
22
|
+
* should stop being waited for at its old speed, and a mean over everything
|
|
23
|
+
* ever observed never forgets.
|
|
24
|
+
*/
|
|
25
|
+
export const TIMING_WINDOW = 50;
|
|
26
|
+
/** Below this many samples an edge has no distribution worth trusting. */
|
|
27
|
+
export const COLD_SAMPLES = 5;
|
|
28
|
+
/**
|
|
29
|
+
* What a cold edge waits: exactly what every step waited before Phase 11.
|
|
30
|
+
*
|
|
31
|
+
* Deliberately unchanged, so the first traversal of an edge behaves as it
|
|
32
|
+
* always did and only a *measured* edge gets a tighter bound. A conservative
|
|
33
|
+
* default that is also the historical default cannot make anything worse.
|
|
34
|
+
*/
|
|
35
|
+
export const COLD_TIMEOUT_MS = 8000;
|
|
36
|
+
/**
|
|
37
|
+
* The hard cap on waiting, from research §7: Nielsen's attention limit. Past
|
|
38
|
+
* ten seconds a person has stopped believing the screen is coming, and so
|
|
39
|
+
* should the agent — it escalates instead.
|
|
40
|
+
*/
|
|
41
|
+
export const HARD_CAP_MS = 10_000;
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* How long to wait for a transition that has been measured.
|
|
45
|
+
*
|
|
46
|
+
* p95 plus a margin, where the margin is the larger of 150 ms and a fifth of
|
|
47
|
+
* p95. The floor matters for fast edges: a tab switch with a p95 of 90 ms
|
|
48
|
+
* would otherwise get a 108 ms budget, and one slow frame would call a
|
|
49
|
+
* perfectly ordinary transition a timeout.
|
|
50
|
+
*/
|
|
51
|
+
export function adaptiveTimeout({ p95, samples } = {}) {
|
|
52
|
+
if (!Number.isFinite(p95) || !Number.isFinite(samples) || samples < COLD_SAMPLES) {
|
|
53
|
+
return { timeoutMs: COLD_TIMEOUT_MS, cold: true, reason: `fewer than ${COLD_SAMPLES} samples` };
|
|
54
|
+
}
|
|
55
|
+
const margin = Math.max(150, Math.round(p95 * 0.2));
|
|
56
|
+
return { timeoutMs: Math.min(HARD_CAP_MS, p95 + margin), cold: false, reason: null, margin };
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Is this transition taking longer than this edge usually does?
|
|
61
|
+
*
|
|
62
|
+
* Two different answers hide behind a slow transition, and §7 asks for both:
|
|
63
|
+
* a screen that is *working* (a spinner, a load) should be waited for up to
|
|
64
|
+
* the hard cap, while a screen that is doing nothing visible has already
|
|
65
|
+
* given its answer. The classifier's `loading` kind is what separates them.
|
|
66
|
+
*/
|
|
67
|
+
export function slowerThanUsual({ elapsedMs, p95, settled, kind } = {}) {
|
|
68
|
+
if (settled || !Number.isFinite(elapsedMs) || !Number.isFinite(p95)) return { slower: false };
|
|
69
|
+
if (elapsedMs <= p95) return { slower: false };
|
|
70
|
+
const working = kind === 'loading';
|
|
71
|
+
return {
|
|
72
|
+
slower: true,
|
|
73
|
+
working,
|
|
74
|
+
keepWaiting: working && elapsedMs < HARD_CAP_MS,
|
|
75
|
+
note: working
|
|
76
|
+
? `slower than usual (${elapsedMs}ms against a p95 of ${p95}ms) and still loading`
|
|
77
|
+
: `slower than usual (${elapsedMs}ms against a p95 of ${p95}ms) with nothing visibly happening`,
|
|
78
|
+
};
|
|
79
|
+
}
|
|
80
|
+
|
|
17
81
|
/**
|
|
18
82
|
* Which fingerprint produced the hashes in these files.
|
|
19
83
|
*
|
|
@@ -276,7 +340,100 @@ export function findScreen(udid, query) {
|
|
|
276
340
|
}
|
|
277
341
|
|
|
278
342
|
/** Remember that doing `action` on `from` led to `to`. */
|
|
279
|
-
|
|
343
|
+
/**
|
|
344
|
+
* Add one observed settle duration to an edge's rolling window.
|
|
345
|
+
*
|
|
346
|
+
* Kept on the edge rather than in a separate store because it is a property of
|
|
347
|
+
* this transition on this screen — the same tap costs 90 ms on a tab bar and
|
|
348
|
+
* 2.4 s on a screen that fetches — and because the graph is already persisted,
|
|
349
|
+
* versioned and pruned.
|
|
350
|
+
*/
|
|
351
|
+
function noteSettle(edge, settleMs, quietGapMs) {
|
|
352
|
+
if (Number.isFinite(settleMs) && settleMs >= 0) {
|
|
353
|
+
edge.settles = [...(edge.settles ?? []), Math.round(settleMs)].slice(-TIMING_WINDOW);
|
|
354
|
+
}
|
|
355
|
+
// Recorded even when zero: "this transition never paused" is exactly the
|
|
356
|
+
// observation that lets the next one stop waiting 500ms to find out.
|
|
357
|
+
if (Number.isFinite(quietGapMs) && quietGapMs >= 0) {
|
|
358
|
+
edge.quietGaps = [...(edge.quietGaps ?? []), Math.round(quietGapMs)].slice(-TIMING_WINDOW);
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
/**
|
|
363
|
+
* How long a screen must hold still on this edge before it is finished.
|
|
364
|
+
*
|
|
365
|
+
* Derived from the longest pause ever seen *inside* this transition, plus a
|
|
366
|
+
* margin, and never longer than the caller's own default — this can only make
|
|
367
|
+
* a wait shorter, never longer, which is what keeps a learned number from
|
|
368
|
+
* becoming a new way to hang.
|
|
369
|
+
*
|
|
370
|
+
* The floor is 150 ms because a settle also needs at least one fresh frame to
|
|
371
|
+
* judge, and the capture loop's own idle interval is the limit on how fast an
|
|
372
|
+
* answer can arrive.
|
|
373
|
+
*/
|
|
374
|
+
export const STILLNESS_FLOOR_MS = 150;
|
|
375
|
+
|
|
376
|
+
export function stillnessFor({ gapSamples, gapP95 } = {}, fallbackMs) {
|
|
377
|
+
if (!Number.isFinite(gapP95) || !Number.isFinite(gapSamples) || gapSamples < COLD_SAMPLES) {
|
|
378
|
+
return { stillnessMs: fallbackMs, cold: true };
|
|
379
|
+
}
|
|
380
|
+
const margin = Math.max(100, Math.round(gapP95 * 0.5));
|
|
381
|
+
return {
|
|
382
|
+
stillnessMs: Math.max(STILLNESS_FLOOR_MS, Math.min(fallbackMs, gapP95 + margin)),
|
|
383
|
+
cold: false,
|
|
384
|
+
};
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
/** What this edge's observed settle durations say, or that it has none. */
|
|
388
|
+
export function timingOf(edge) {
|
|
389
|
+
const samples = edge?.settles ?? [];
|
|
390
|
+
const gaps = edge?.quietGaps ?? [];
|
|
391
|
+
return {
|
|
392
|
+
samples: samples.length,
|
|
393
|
+
p50: metrics.percentile(samples, 50),
|
|
394
|
+
p95: metrics.percentile(samples, 95),
|
|
395
|
+
gapSamples: gaps.length,
|
|
396
|
+
gapP95: metrics.percentile(gaps, 95),
|
|
397
|
+
};
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
/**
|
|
401
|
+
* How long to wait for `step` on `screen`, from what it has cost before.
|
|
402
|
+
*
|
|
403
|
+
* Returns the cold default when this screen or this action has not been
|
|
404
|
+
* measured, and says which — a timeout nobody can explain is how a fixed sleep
|
|
405
|
+
* gets reintroduced as a constant with a comment.
|
|
406
|
+
*/
|
|
407
|
+
export function timingFor(udid, screen, step) {
|
|
408
|
+
const node = screen?.hash ? nearestScreen(udid, screen)?.node : null;
|
|
409
|
+
const edge = node?.edges?.find((e) => e.action === actionSignature(step));
|
|
410
|
+
const stats = timingOf(edge);
|
|
411
|
+
return { ...stats, ...adaptiveTimeout(stats), known: Boolean(edge) };
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
/**
|
|
415
|
+
* What getting *to* this screen has cost before.
|
|
416
|
+
*
|
|
417
|
+
* `sim_state` is asked "what is on screen and is it done moving", with no
|
|
418
|
+
* action in hand, so there is no outgoing edge to consult. The useful answer
|
|
419
|
+
* is the inbound one: the edge most recently traversed into this screen is how
|
|
420
|
+
* we got here, and its distribution is what "slower than usual" means right
|
|
421
|
+
* now. Most recently seen rather than most travelled — a screen reachable two
|
|
422
|
+
* ways is being timed against the way it was just reached.
|
|
423
|
+
*/
|
|
424
|
+
export function timingInto(udid, hash) {
|
|
425
|
+
if (!hash) return null;
|
|
426
|
+
let best = null;
|
|
427
|
+
for (const node of allNodes(udid)) {
|
|
428
|
+
for (const edge of node.edges ?? []) {
|
|
429
|
+
if (edge.to !== hash) continue;
|
|
430
|
+
if (!best || (edge.lastSeen ?? 0) > (best.lastSeen ?? 0)) best = edge;
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
return best ? { ...timingOf(best), action: best.action, kind: best.kind ?? null } : null;
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
export function record(udid, { from, action, to, kind, settleMs, quietGapMs }) {
|
|
280
437
|
const fromKey = typeof from === 'string' ? { hash: from } : from;
|
|
281
438
|
const toHash = typeof to === 'string' ? to : to?.hash;
|
|
282
439
|
if (!fromKey?.hash || !toHash) return null;
|
|
@@ -343,6 +500,7 @@ export function record(udid, { from, action, to, kind }) {
|
|
|
343
500
|
save(udid, target);
|
|
344
501
|
existing.count += 1;
|
|
345
502
|
existing.lastSeen = Date.now();
|
|
503
|
+
noteSettle(existing, settleMs, quietGapMs);
|
|
346
504
|
save(udid, node);
|
|
347
505
|
return node;
|
|
348
506
|
}
|
|
@@ -354,6 +512,7 @@ export function record(udid, { from, action, to, kind }) {
|
|
|
354
512
|
existing.kind = kind ?? existing.kind;
|
|
355
513
|
existing.count += 1;
|
|
356
514
|
existing.lastSeen = Date.now();
|
|
515
|
+
noteSettle(existing, settleMs);
|
|
357
516
|
} else {
|
|
358
517
|
node.edges.push({
|
|
359
518
|
action: signature,
|
|
@@ -364,6 +523,8 @@ export function record(udid, { from, action, to, kind }) {
|
|
|
364
523
|
kind,
|
|
365
524
|
count: 1,
|
|
366
525
|
lastSeen: Date.now(),
|
|
526
|
+
settles: Number.isFinite(settleMs) && settleMs >= 0 ? [Math.round(settleMs)] : [],
|
|
527
|
+
quietGaps: Number.isFinite(quietGapMs) && quietGapMs >= 0 ? [Math.round(quietGapMs)] : [],
|
|
367
528
|
});
|
|
368
529
|
}
|
|
369
530
|
save(udid, node);
|