simframe 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -0
- package/flows/hpi-suite.json +68 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +49 -1
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +7 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +11 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +48 -0
- package/native/simframed/Sources/simframed/main.swift +128 -73
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +40 -0
- package/package.json +2 -1
- package/scripts/bench-hpi.mjs +254 -0
- package/scripts/check-package.mjs +7 -0
- package/src/actions.js +184 -4
- package/src/baseline.js +333 -0
- package/src/cli.js +331 -2
- package/src/daemon.js +9 -0
- package/src/fingerprint.js +7 -1
- package/src/graph.js +162 -1
- package/src/index.js +118 -20
- package/src/input.js +111 -1
- package/src/intent.js +11 -2
- package/src/matching.js +81 -2
- package/src/mcp.js +14 -1
- package/src/metrics.js +499 -0
- package/src/navigate.js +44 -7
- package/src/platform/android.js +15 -0
- package/src/platform/index.js +3 -0
- package/src/platform/ios.js +40 -0
- package/src/screenmap.js +36 -14
- package/src/view.js +4 -3
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// The agent half of the Human Parity Index.
|
|
3
|
+
//
|
|
4
|
+
// Runs every flow in the suite N times, exactly as an agent would run it, and
|
|
5
|
+
// reduces the flow records to an HPI report. The human half is recorded once
|
|
6
|
+
// by a person (`simframe baseline record`) and committed; this side is
|
|
7
|
+
// re-measured on every run, which is what makes HPI a trend rather than a
|
|
8
|
+
// claim.
|
|
9
|
+
//
|
|
10
|
+
// Two rules this file follows, both learned in this repo:
|
|
11
|
+
//
|
|
12
|
+
// - Nothing is verified through a pipe. `node script.mjs | tail -3` exits
|
|
13
|
+
// with tail's status, and this project has already shipped a check that
|
|
14
|
+
// could not fail because of it. Every gate here is an exit code.
|
|
15
|
+
// - A missing human baseline is reported, never defaulted. HPI_time with no
|
|
16
|
+
// denominator is null; it is not 1.0, and it is not "parity".
|
|
17
|
+
import fs from 'node:fs';
|
|
18
|
+
import path from 'node:path';
|
|
19
|
+
import { fileURLToPath } from 'node:url';
|
|
20
|
+
import { runScript } from '../src/actions.js';
|
|
21
|
+
import * as api from '../src/index.js';
|
|
22
|
+
import * as baseline from '../src/baseline.js';
|
|
23
|
+
import * as metrics from '../src/metrics.js';
|
|
24
|
+
|
|
25
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
26
|
+
const arg = (name, fallback = null) => {
|
|
27
|
+
const hit = process.argv.find((a) => a.startsWith(`--${name}=`));
|
|
28
|
+
return hit ? hit.slice(name.length + 3) : fallback;
|
|
29
|
+
};
|
|
30
|
+
const has = (name) => process.argv.includes(`--${name}`);
|
|
31
|
+
|
|
32
|
+
const device = arg('device');
|
|
33
|
+
const runs = Math.max(1, Number(arg('runs', '3')));
|
|
34
|
+
const only = arg('flow');
|
|
35
|
+
const out = arg('out');
|
|
36
|
+
const baselineFile = arg('baseline', path.join(ROOT, 'docs', 'research', 'hpi-baseline.json'));
|
|
37
|
+
const gate = has('gate');
|
|
38
|
+
/**
|
|
39
|
+
* How many times to measure the whole suite.
|
|
40
|
+
*
|
|
41
|
+
* Three when gating, because one measurement is mostly noise: identical code
|
|
42
|
+
* measured three times the same afternoon gave HPI_time 0.475, 0.413 and
|
|
43
|
+
* 0.371. The gate reads the median of the passes, which is what lets the time
|
|
44
|
+
* band stay as tight as 25% instead of being widened to cover a single run's
|
|
45
|
+
* spread.
|
|
46
|
+
*/
|
|
47
|
+
const passes = Math.max(1, Number(arg('passes', gate ? '3' : '1')));
|
|
48
|
+
/**
|
|
49
|
+
* A pause between runs, and it is not superstition.
|
|
50
|
+
*
|
|
51
|
+
* Measured on one 8-run loop: 15.2 s, 15.2 s, then 6.9 s failing, then 25.7 s,
|
|
52
|
+
* 25.7 s, then the display wedged — and it wedged again within a couple of
|
|
53
|
+
* minutes of the same load after a device restart. Relaunching an app as fast
|
|
54
|
+
* as a script can is not what this suite is trying to measure, and a simulator
|
|
55
|
+
* asked to do it degrades and then stops rendering. The suite paces itself so
|
|
56
|
+
* the numbers describe simframe rather than the simulator's tolerance for
|
|
57
|
+
* being hammered.
|
|
58
|
+
*/
|
|
59
|
+
const cooldownMs = Math.max(0, Number(arg('cooldown', '1500')));
|
|
60
|
+
|
|
61
|
+
const suite = baseline.loadSuite(arg('suite', baseline.SUITE_FILE)).filter((f) => !only || f.name === only);
|
|
62
|
+
if (!suite.length) {
|
|
63
|
+
console.error(`no flows to run${only ? ` matching "${only}"` : ''}`);
|
|
64
|
+
process.exit(2);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
const { device: dev } = await api.ensureDaemon(device);
|
|
68
|
+
console.log(`device: ${dev.name} (${dev.udid})`);
|
|
69
|
+
console.log(`suite: ${suite.map((f) => f.name).join(', ')} × ${runs} run(s) × ${passes} pass(es)\n`);
|
|
70
|
+
|
|
71
|
+
const ids = new Set();
|
|
72
|
+
let hardFailures = 0;
|
|
73
|
+
/**
|
|
74
|
+
* A wedged capture loop is not a slow flow, and every run after it is doomed.
|
|
75
|
+
*
|
|
76
|
+
* Measured four times in one session: the display surface stops being readable
|
|
77
|
+
* mid-suite, the daemon re-resolves the display port and still gets nothing,
|
|
78
|
+
* and only restarting the device cures it. Grinding through the remaining runs
|
|
79
|
+
* produced three identical "could not run at all" lines and an HPI computed
|
|
80
|
+
* from whatever happened to finish first — a number with a hole in it, which
|
|
81
|
+
* is worse than no number.
|
|
82
|
+
*/
|
|
83
|
+
const WEDGED = /display surface could not be read|did not produce a frame/;
|
|
84
|
+
/**
|
|
85
|
+
* The host could not do the thing, as distinct from simframe doing it slowly.
|
|
86
|
+
*
|
|
87
|
+
* A hosted runner takes 47-55 s to fail `simctl launch com.apple.Preferences`
|
|
88
|
+
* and then fails it again, three runs in a row — the same class of fault this
|
|
89
|
+
* repo already records for `simctl openurl`, which returns "Operation timed
|
|
90
|
+
* out" on a loaded runner. Six of those is nine minutes of CI spent measuring
|
|
91
|
+
* the runner's patience, and the resulting HPI describes nothing.
|
|
92
|
+
*/
|
|
93
|
+
const ENVIRONMENT = /could not launch|Command failed: xcrun simctl (launch|terminate)|Operation timed out|timed out/i;
|
|
94
|
+
/** Two environmental failures of the same flow is the environment, not a flake. */
|
|
95
|
+
const ENV_GIVE_UP = 2;
|
|
96
|
+
let abort = null;
|
|
97
|
+
const envFailures = new Map();
|
|
98
|
+
|
|
99
|
+
const passSets = [];
|
|
100
|
+
outer: for (let pass = 1; pass <= passes; pass += 1) {
|
|
101
|
+
const thisPass = new Set();
|
|
102
|
+
passSets.push(thisPass);
|
|
103
|
+
if (passes > 1) console.log(`pass ${pass}/${passes}`);
|
|
104
|
+
for (const flow of suite) {
|
|
105
|
+
for (let run = 1; run <= runs; run += 1) {
|
|
106
|
+
// The same start state the human baseline was recorded from: app not
|
|
107
|
+
// running, on the home screen. Not part of the timed flow, and
|
|
108
|
+
// deliberately not expressed as flow steps — see baseline.resetFor.
|
|
109
|
+
// The reset and its settle live inside the same guard as the run.
|
|
110
|
+
//
|
|
111
|
+
// They did not, and a device whose display had failed threw out of
|
|
112
|
+
// `waitFor` — outside the try — killing the process with a stack trace
|
|
113
|
+
// instead of the one classification this script exists to make. An
|
|
114
|
+
// environmental failure has to be classified wherever it happens, not
|
|
115
|
+
// only where it was convenient to catch.
|
|
116
|
+
let res;
|
|
117
|
+
try {
|
|
118
|
+
const reset = await baseline.resetFor(dev.udid, flow);
|
|
119
|
+
if (reset.failures.length) console.log(` (reset: ${reset.failures.join('; ')})`);
|
|
120
|
+
await api.waitFor(dev.udid, { mode: 'stable', stableMs: 400, timeoutMs: 4000 });
|
|
121
|
+
res = await runScript(dev.udid, {
|
|
122
|
+
steps: flow.steps,
|
|
123
|
+
flowName: flow.name,
|
|
124
|
+
minSteps: flow.minSteps ?? null,
|
|
125
|
+
});
|
|
126
|
+
} catch (err) {
|
|
127
|
+
// A flow that could not start at all is not a slow flow. It has no
|
|
128
|
+
// record, so it cannot be averaged into anything; it is reported and
|
|
129
|
+
// counted.
|
|
130
|
+
hardFailures += 1;
|
|
131
|
+
console.log(`FAIL ${flow.name} run ${run}: ${err.message}`);
|
|
132
|
+
if (WEDGED.test(err.message)) {
|
|
133
|
+
abort = { kind: 'capture is wedged', message: err.message };
|
|
134
|
+
break outer;
|
|
135
|
+
}
|
|
136
|
+
if (ENVIRONMENT.test(err.message)) {
|
|
137
|
+
const n = (envFailures.get(flow.name) ?? 0) + 1;
|
|
138
|
+
envFailures.set(flow.name, n);
|
|
139
|
+
if (n >= ENV_GIVE_UP) {
|
|
140
|
+
abort = { kind: `the host cannot run "${flow.name}"`, message: err.message.split('\n')[0] };
|
|
141
|
+
break outer;
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
ids.add(res.flowId);
|
|
147
|
+
thisPass.add(res.flowId);
|
|
148
|
+
const verdicts = res.results.map((r) => r.verification?.verdict).filter(Boolean);
|
|
149
|
+
console.log(
|
|
150
|
+
`${res.ok ? 'ok ' : 'FAIL'} ${flow.name.padEnd(24)} run ${run}/${runs} ` +
|
|
151
|
+
`${String(res.totalMs).padStart(6)}ms ${res.ranSteps}/${res.totalSteps} steps ` +
|
|
152
|
+
`${verdicts.filter((v) => v !== 'ok').length ? `verdicts: ${verdicts.join(',')}` : 'all ok'}`,
|
|
153
|
+
);
|
|
154
|
+
if (!res.ok) console.log(` ${res.results.filter((r) => !r.ok).map((r) => r.error).join('; ')}`);
|
|
155
|
+
if (cooldownMs) await new Promise((r) => setTimeout(r, cooldownMs));
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// This process's runs only. The log is append-only and holds every earlier
|
|
161
|
+
// measurement, which is what makes it a trend — but a CI number computed over
|
|
162
|
+
// somebody's local runs from last week is not this commit's number.
|
|
163
|
+
const flows = metrics.readFlows(dev.udid).filter((f) => ids.has(f.flow_id));
|
|
164
|
+
const humans = baseline.readBaselines();
|
|
165
|
+
// Each pass measured on its own, so the gate can take a median over them
|
|
166
|
+
// rather than trusting one. Accuracy is pooled over every run instead: one
|
|
167
|
+
// wrong action in thirty is a wrong action, and a median would hide it.
|
|
168
|
+
const allFlows = metrics.readFlows(dev.udid);
|
|
169
|
+
const passReports = passSets
|
|
170
|
+
.map((set) => metrics.hpi({ flows: allFlows.filter((f) => set.has(f.flow_id)), baselines: humans }))
|
|
171
|
+
.filter((r) => r.overall.runs > 0);
|
|
172
|
+
const passTimes = passReports.map((r) => r.overall.hpi_time).filter((t) => Number.isFinite(t));
|
|
173
|
+
|
|
174
|
+
const report = {
|
|
175
|
+
...metrics.hpi({ flows, baselines: humans }),
|
|
176
|
+
measured_at: new Date().toISOString(),
|
|
177
|
+
device: { udid: dev.udid, name: dev.name, runtime: dev.runtime },
|
|
178
|
+
runs_per_flow: runs,
|
|
179
|
+
passes,
|
|
180
|
+
pass_hpi_time: passTimes,
|
|
181
|
+
hard_failures: hardFailures,
|
|
182
|
+
human_baselines: Object.fromEntries(
|
|
183
|
+
Object.entries(humans).map(([k, v]) => [k, { runs: v.runs, p50: v.wall_time_ms?.p50 ?? null }]),
|
|
184
|
+
),
|
|
185
|
+
};
|
|
186
|
+
|
|
187
|
+
console.log('\nflow runs agent p50 human p50 HPI_time step_ratio');
|
|
188
|
+
for (const f of report.flows) {
|
|
189
|
+
console.log(
|
|
190
|
+
`${f.flow.padEnd(24)} ${String(f.runs).padStart(5)} ${`${f.agent_ms.p50}ms`.padStart(9)} ` +
|
|
191
|
+
`${(f.human_median_ms ? `${f.human_median_ms}ms` : '—').padStart(9)} ` +
|
|
192
|
+
`${String(f.hpi_time ?? '—').padStart(8)} ${String(f.step_ratio ?? '—').padStart(10)}`,
|
|
193
|
+
);
|
|
194
|
+
}
|
|
195
|
+
report.overall.hpi_time_median_of_passes = passTimes.length ? Number(metrics.median(passTimes).toFixed(3)) : null;
|
|
196
|
+
const o = report.overall;
|
|
197
|
+
console.log(`\nHPI_accuracy ${o.hpi_accuracy} HPI_time ${o.hpi_time ?? '—'} HPI ${o.hpi ?? '—'} step_ratio ${o.step_ratio ?? '—'}`);
|
|
198
|
+
if (passTimes.length > 1) {
|
|
199
|
+
console.log(`HPI_time per pass: ${passTimes.join(', ')} — median ${o.hpi_time_median_of_passes} (what the gate reads)`);
|
|
200
|
+
}
|
|
201
|
+
if (o.hpi_time == null) {
|
|
202
|
+
console.log(`no human baseline for any measured flow — HPI_time and HPI are null, not 1.0.`);
|
|
203
|
+
console.log(`record one: simframe baseline record <flow> --device=${dev.udid} --runs=5`);
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
if (out) {
|
|
207
|
+
fs.writeFileSync(out, `${JSON.stringify(report, null, 2)}\n`);
|
|
208
|
+
console.log(`wrote ${out}`);
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
const escalations = metrics.breakdown(metrics.readEscalations(dev.udid));
|
|
212
|
+
console.log(`\nescalations in this device's log: ${escalations.total} (${Object.entries(escalations.by_reason).filter(([, n]) => n).map(([r, n]) => `${r} ${n}`).join(', ') || 'none'})`);
|
|
213
|
+
|
|
214
|
+
if (abort) {
|
|
215
|
+
console.error(`\n${abort.kind}: ${abort.message}`);
|
|
216
|
+
console.error('No comparable HPI was measured — what is above is a partial suite and');
|
|
217
|
+
console.error('must not be adopted as a baseline or read as a regression.');
|
|
218
|
+
if (/wedged/.test(abort.kind)) {
|
|
219
|
+
console.error('Only restarting the device is known to cure a wedge:');
|
|
220
|
+
console.error(` xcrun simctl shutdown ${dev.udid} && xcrun simctl boot ${dev.udid}`);
|
|
221
|
+
}
|
|
222
|
+
// 2, not 1. "I could not measure" and "it got worse" are different answers,
|
|
223
|
+
// and a job that reports them with the same exit code teaches people to
|
|
224
|
+
// ignore both. The workflow treats 2 as a loud warning and 1 as a failure.
|
|
225
|
+
process.exit(2);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
if (hardFailures) {
|
|
229
|
+
console.error(`\n${hardFailures} flow run(s) could not run at all.`);
|
|
230
|
+
process.exit(1);
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
if (!gate) process.exit(0);
|
|
234
|
+
|
|
235
|
+
// ------------------------------------------------------------------ the gate
|
|
236
|
+
const committed = fs.existsSync(baselineFile) ? JSON.parse(fs.readFileSync(baselineFile, 'utf8')) : null;
|
|
237
|
+
if (!committed) {
|
|
238
|
+
// Loudly inactive rather than quietly passing. A gate with nothing to
|
|
239
|
+
// compare against cannot detect a regression, and saying "ok" here is how a
|
|
240
|
+
// CI job comes to mean nothing.
|
|
241
|
+
console.log(`\nGATE INACTIVE — no committed baseline at ${path.relative(ROOT, baselineFile)}.`);
|
|
242
|
+
console.log('Commit this run as the baseline to arm it.');
|
|
243
|
+
process.exit(0);
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
const base = committed.overall ?? {};
|
|
247
|
+
const failures = metrics.gateAgainst(committed, report);
|
|
248
|
+
|
|
249
|
+
console.log(`\ngate vs ${path.relative(ROOT, baselineFile)} (measured ${committed.measured_at ?? '?'})`);
|
|
250
|
+
console.log(` HPI_accuracy ${base.hpi_accuracy ?? '—'} -> ${o.hpi_accuracy ?? '—'}`);
|
|
251
|
+
console.log(` HPI_time ${metrics.gateTime(base) ?? '—'} -> ${metrics.gateTime(o) ?? '—'}`);
|
|
252
|
+
console.log(` band ${metrics.TIME_REGRESSION * 100}% (median of ${passTimes.length || 1} pass(es))`);
|
|
253
|
+
for (const f of failures) console.log(`FAIL ${f}`);
|
|
254
|
+
process.exit(failures.length ? 1 : 0);
|
|
@@ -75,6 +75,13 @@ function requiredFiles() {
|
|
|
75
75
|
throw new Error('no skill found under skills/ — has the layout moved? This check would silently pass.');
|
|
76
76
|
}
|
|
77
77
|
|
|
78
|
+
// The flow suite. `simframe baseline` and `simframe hpi` read it at
|
|
79
|
+
// runtime, so an absent one is a command that only fails once installed.
|
|
80
|
+
for (const f of walk(path.join(ROOT, 'flows'), (p) => p.endsWith('.json'))) required.add(rel(f));
|
|
81
|
+
if (!walk(path.join(ROOT, 'flows'), (p) => p.endsWith('.json')).length) {
|
|
82
|
+
throw new Error('no flow suite found under flows/ — has the layout moved? This check would silently pass.');
|
|
83
|
+
}
|
|
84
|
+
|
|
78
85
|
// Anything package.json points at to run.
|
|
79
86
|
const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8'));
|
|
80
87
|
for (const target of Object.values(pkg.bin ?? {})) required.add(target.replace(/^\.\//, ''));
|
package/src/actions.js
CHANGED
|
@@ -6,6 +6,8 @@ import * as api from './index.js';
|
|
|
6
6
|
import * as graph from './graph.js';
|
|
7
7
|
import * as input from './input.js';
|
|
8
8
|
import * as intent from './intent.js';
|
|
9
|
+
import * as metrics from './metrics.js';
|
|
10
|
+
import * as screenmap from './screenmap.js';
|
|
9
11
|
import { launchApp, openUrl, setPermission, terminateApp } from './platform/index.js';
|
|
10
12
|
|
|
11
13
|
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
@@ -112,6 +114,12 @@ export async function runScript(
|
|
|
112
114
|
// Rebuild the HID session and retry once when a hardware button provably
|
|
113
115
|
// did nothing. Off only for a caller deliberately testing that path.
|
|
114
116
|
recoverInput = true,
|
|
117
|
+
// What this run is called and how few steps it could take, for the flow
|
|
118
|
+
// record. A bare `sim_do` has neither and says so with nulls rather than
|
|
119
|
+
// inventing a name — an unnamed run still gets timed, it just cannot be
|
|
120
|
+
// compared against a human baseline.
|
|
121
|
+
flowName = null,
|
|
122
|
+
minSteps = null,
|
|
115
123
|
options,
|
|
116
124
|
} = {},
|
|
117
125
|
) {
|
|
@@ -119,6 +127,24 @@ export async function runScript(
|
|
|
119
127
|
const { device } = await api.ensureDaemon(deviceQuery, options);
|
|
120
128
|
const udid = device.udid;
|
|
121
129
|
const startedAt = Date.now();
|
|
130
|
+
// Measurement only. Nothing below reads these, and a failure to write one
|
|
131
|
+
// can never change what a step does — see `note`.
|
|
132
|
+
const flowId = metrics.newFlowId();
|
|
133
|
+
const escalations = [];
|
|
134
|
+
const verdicts = [];
|
|
135
|
+
// Named noteEscalation, not note: the step loop below declares its own
|
|
136
|
+
// `note` string for the no-visible-change suffix, which shadowed this and
|
|
137
|
+
// turned every escalating verdict into a failed step reading "note is not a
|
|
138
|
+
// function". The try/catch inside here could not help — the throw was at the
|
|
139
|
+
// call site, one scope out. Instrumentation that can fail a flow is worse
|
|
140
|
+
// than no instrumentation.
|
|
141
|
+
const noteEscalation = (record) => {
|
|
142
|
+
try {
|
|
143
|
+
escalations.push(metrics.recordEscalation(udid, { flowId, ...record }));
|
|
144
|
+
} catch {
|
|
145
|
+
/* instrumentation must not be able to fail a flow it is only watching */
|
|
146
|
+
}
|
|
147
|
+
};
|
|
122
148
|
|
|
123
149
|
const needsInput = steps.some((s) => ACTION_STEPS.has(normalizeStep(s).action));
|
|
124
150
|
if (needsInput) {
|
|
@@ -165,13 +191,58 @@ export async function runScript(
|
|
|
165
191
|
const prediction = verify && beforeScreen?.hash ? graph.predict(udid, beforeScreen, step) : null;
|
|
166
192
|
try {
|
|
167
193
|
let detail = await runStep(deviceQuery, udid, step, { screen, options, frames });
|
|
194
|
+
// How long this transition has cost before, on this screen, for this
|
|
195
|
+
// action. A cold edge gets the old fixed default and says so; a measured
|
|
196
|
+
// one gets p95 plus a margin. Research §7.
|
|
197
|
+
const learned = verify && beforeScreen?.hash ? graph.timingFor(udid, beforeScreen, step) : null;
|
|
198
|
+
// How long this screen must hold still before it counts as settled.
|
|
199
|
+
//
|
|
200
|
+
// 500 ms was a constant paid by every step of every flow, and it is the
|
|
201
|
+
// reducible half of a settle: the rest is the transition genuinely
|
|
202
|
+
// taking time. An edge whose transition has never paused mid-flight
|
|
203
|
+
// needs 150 ms of quiet, not 500. Capped at the caller's own value, so
|
|
204
|
+
// this can only ever shorten a wait.
|
|
205
|
+
// NOT YET USED TO DECIDE ANYTHING, and the reason is worth the space.
|
|
206
|
+
//
|
|
207
|
+
// `graph.stillnessFor` computes a shorter window from the longest pause
|
|
208
|
+
// ever observed inside this transition, and measured live it made the
|
|
209
|
+
// Settings flow 8.0 s instead of 11.5 s — and wrong. Eight runs in a row
|
|
210
|
+
// failed at step 2 with the screen still showing Settings root, because
|
|
211
|
+
// step 1's settle returned mid-push, `screenIdentity` then read the
|
|
212
|
+
// screen we had not left yet, and the graph learned root -> root as a
|
|
213
|
+
// verified edge and started predicting it.
|
|
214
|
+
//
|
|
215
|
+
// The flaw is in the estimator, not the idea: the gap statistic is
|
|
216
|
+
// gathered only from what a wait itself observed, so a wait that ends
|
|
217
|
+
// early never sees the pauses that come later, the gaps look like zero,
|
|
218
|
+
// the window ratchets down, and the next wait ends earlier still. A
|
|
219
|
+
// self-reinforcing bias with a wrong graph at the end of it.
|
|
220
|
+
//
|
|
221
|
+
// The unbiased estimator is available and is a separate piece of work:
|
|
222
|
+
// the frame history holds every frame's timestamp and diff, so the true
|
|
223
|
+
// motion profile of a transition can be computed *after* it is over
|
|
224
|
+
// rather than from inside the wait that cut it short. Until then the
|
|
225
|
+
// gaps are recorded and not acted on — measuring is safe, and this is
|
|
226
|
+
// Phase 11's own rule that a learned number may only ever shorten a
|
|
227
|
+
// wait, applied to itself.
|
|
228
|
+
const stillness = step.stableMs ?? stableMs;
|
|
229
|
+
const stillnessPlan = learned ? graph.stillnessFor(learned, stillness) : { cold: true };
|
|
230
|
+
// A settle is not satisfied until the screen has held still for
|
|
231
|
+
// `stillness`, so a budget below that can never be met — and the learned
|
|
232
|
+
// p95 is measured from waits that include the stillness window, which
|
|
233
|
+
// makes it self-consistent but not self-evidently so. A tab switch with
|
|
234
|
+
// a p95 of 90ms would get a 240ms budget and then time out at 240ms
|
|
235
|
+
// waiting for 500ms of quiet, turning every fast edge into a failure.
|
|
236
|
+
const floorMs = stillness + 250;
|
|
237
|
+
const budgetMs = step.timeoutMs
|
|
238
|
+
?? (learned && !learned.cold ? Math.max(learned.timeoutMs, floorMs) : timeoutMs);
|
|
168
239
|
const settleFor = async () => {
|
|
169
240
|
if (!autoSettle || !ACTION_STEPS.has(step.action)) return null;
|
|
170
241
|
const w = await api.waitFor(deviceQuery, {
|
|
171
242
|
mode: 'settle',
|
|
172
243
|
since: before,
|
|
173
|
-
stableMs:
|
|
174
|
-
timeoutMs:
|
|
244
|
+
stableMs: stillness,
|
|
245
|
+
timeoutMs: budgetMs,
|
|
175
246
|
options,
|
|
176
247
|
});
|
|
177
248
|
return {
|
|
@@ -180,9 +251,55 @@ export async function runScript(
|
|
|
180
251
|
sawChange: w.sawChange,
|
|
181
252
|
stalled: Boolean(w.stalled),
|
|
182
253
|
noVisibleChange: Boolean(w.noVisibleChange),
|
|
254
|
+
// What this wait was allowed, and where the number came from. A
|
|
255
|
+
// timeout nobody can explain is how a fixed sleep comes back as a
|
|
256
|
+
// constant with a comment.
|
|
257
|
+
budgetMs,
|
|
258
|
+
stillnessMs: stillness,
|
|
259
|
+
quietGapMs: w.quietGapMs,
|
|
260
|
+
timing: learned
|
|
261
|
+
? {
|
|
262
|
+
p50: learned.p50,
|
|
263
|
+
p95: learned.p95,
|
|
264
|
+
samples: learned.samples,
|
|
265
|
+
cold: learned.cold,
|
|
266
|
+
gapP95: learned.gapP95,
|
|
267
|
+
gapSamples: learned.gapSamples,
|
|
268
|
+
// What it *would* have been, for the eval that has to happen
|
|
269
|
+
// before this is trusted with a wait.
|
|
270
|
+
stillnessWouldBe: stillnessPlan.stillnessMs ?? null,
|
|
271
|
+
}
|
|
272
|
+
: null,
|
|
183
273
|
};
|
|
184
274
|
};
|
|
185
275
|
let settled = await settleFor();
|
|
276
|
+
// A screen that is still working earns more time; a screen doing nothing
|
|
277
|
+
// visible has already answered. Research §7: keep waiting past p95 only
|
|
278
|
+
// while the transition classifier says something is loading, and never
|
|
279
|
+
// past Nielsen's 10 s — at which point it escalates with the timing
|
|
280
|
+
// attached rather than waiting longer.
|
|
281
|
+
if (settled && !settled.ok && !settled.noVisibleChange && learned && !learned.cold) {
|
|
282
|
+
const kind = (await api.getState(deviceQuery, { options })).state.transition?.kind;
|
|
283
|
+
const verdict = graph.slowerThanUsual({
|
|
284
|
+
elapsedMs: settled.waitedMs, p95: learned.p95, settled: false, kind,
|
|
285
|
+
});
|
|
286
|
+
if (verdict.keepWaiting) {
|
|
287
|
+
const remaining = graph.HARD_CAP_MS - settled.waitedMs;
|
|
288
|
+
const more = await api.waitFor(deviceQuery, {
|
|
289
|
+
mode: 'settle', since: before, stableMs: stillness, timeoutMs: remaining, options,
|
|
290
|
+
});
|
|
291
|
+
settled = {
|
|
292
|
+
...settled,
|
|
293
|
+
ok: more.satisfied,
|
|
294
|
+
waitedMs: settled.waitedMs + more.waitedMs,
|
|
295
|
+
sawChange: settled.sawChange || more.sawChange,
|
|
296
|
+
quietGapMs: Math.max(settled.quietGapMs ?? 0, more.quietGapMs ?? 0),
|
|
297
|
+
slowerThanUsual: verdict.note,
|
|
298
|
+
};
|
|
299
|
+
} else if (verdict.slower) {
|
|
300
|
+
settled = { ...settled, slowerThanUsual: verdict.note };
|
|
301
|
+
}
|
|
302
|
+
}
|
|
186
303
|
|
|
187
304
|
// A hardware button that moved nothing did not arrive.
|
|
188
305
|
//
|
|
@@ -227,7 +344,19 @@ export async function runScript(
|
|
|
227
344
|
// nothing at all.
|
|
228
345
|
endScreen = afterScreen;
|
|
229
346
|
if (afterScreen.confirmed && afterScreen.hash) {
|
|
230
|
-
|
|
347
|
+
// The observed cost of this transition, which is what makes the next
|
|
348
|
+
// one adaptive. Only from a settle that was actually satisfied: a
|
|
349
|
+
// timeout is not a measurement of how long the screen takes, it is a
|
|
350
|
+
// measurement of how long we were prepared to wait.
|
|
351
|
+
graph.record(udid, {
|
|
352
|
+
from: beforeScreen, action: step, to: afterScreen, kind,
|
|
353
|
+
settleMs: settled?.ok ? settled.waitedMs : undefined,
|
|
354
|
+
// The pause statistic is worth having from any settle that saw the
|
|
355
|
+
// screen move, satisfied or not: a transition that paused for
|
|
356
|
+
// 400ms and then timed out is exactly the case a 150ms stillness
|
|
357
|
+
// window would have got wrong.
|
|
358
|
+
quietGapMs: settled?.sawChange ? settled.quietGapMs : undefined,
|
|
359
|
+
});
|
|
231
360
|
carriedScreen = afterScreen;
|
|
232
361
|
}
|
|
233
362
|
}
|
|
@@ -244,6 +373,25 @@ export async function runScript(
|
|
|
244
373
|
settled,
|
|
245
374
|
});
|
|
246
375
|
const halt = haltDecision({ verification, stopOnUnexpected, continueOnError });
|
|
376
|
+
if (verification?.verdict) verdicts.push(verification.verdict);
|
|
377
|
+
if (metrics.ESCALATING_VERDICTS.has(verification?.verdict)) {
|
|
378
|
+
noteEscalation({
|
|
379
|
+
stepIndex: i,
|
|
380
|
+
fingerprint: beforeScreen?.hash ?? null,
|
|
381
|
+
reason: 'verification_failed',
|
|
382
|
+
candidates: [],
|
|
383
|
+
// A halted run is a decision simframe made and stopped on; a step
|
|
384
|
+
// that moved nothing carries on and leaves the judgement to whoever
|
|
385
|
+
// reads the result.
|
|
386
|
+
outcome: halt.halt ? 'failed' : 'escalated_to_model',
|
|
387
|
+
wallMs: Date.now() - stepStart,
|
|
388
|
+
detail: `${verification.verdict}: ${verification.detail}`
|
|
389
|
+
+ (settled?.slowerThanUsual ? ` [${settled.slowerThanUsual}]` : '')
|
|
390
|
+
+ (settled?.timing && !settled.timing.cold
|
|
391
|
+
? ` [waited ${settled.waitedMs}ms of a ${settled.budgetMs}ms budget; p95 ${settled.timing.p95}ms]`
|
|
392
|
+
: ''),
|
|
393
|
+
});
|
|
394
|
+
}
|
|
247
395
|
if (halt.halt) {
|
|
248
396
|
results[results.length - 1].ok = false;
|
|
249
397
|
results[results.length - 1].error = halt.error;
|
|
@@ -252,20 +400,52 @@ export async function runScript(
|
|
|
252
400
|
}
|
|
253
401
|
} catch (err) {
|
|
254
402
|
results.push({ index: i, action: step.action, ok: false, ms: Date.now() - stepStart, error: err.message });
|
|
403
|
+
const why = metrics.reasonForStepError(step, err);
|
|
404
|
+
noteEscalation({
|
|
405
|
+
stepIndex: i,
|
|
406
|
+
fingerprint: beforeScreen?.hash ?? metrics.fingerprintNow(udid, screenmap),
|
|
407
|
+
reason: why.reason,
|
|
408
|
+
candidates: why.candidates,
|
|
409
|
+
tried: why.tried,
|
|
410
|
+
outcome: 'failed',
|
|
411
|
+
wallMs: Date.now() - stepStart,
|
|
412
|
+
detail: err.message,
|
|
413
|
+
});
|
|
255
414
|
failed = true;
|
|
256
415
|
if (!continueOnError) break;
|
|
257
416
|
}
|
|
258
417
|
}
|
|
259
418
|
|
|
419
|
+
const wallMs = Date.now() - startedAt;
|
|
420
|
+
try {
|
|
421
|
+
metrics.recordFlow(udid, metrics.flowRecordFrom({
|
|
422
|
+
flowId,
|
|
423
|
+
flowName,
|
|
424
|
+
udid,
|
|
425
|
+
startedAt,
|
|
426
|
+
wallMs,
|
|
427
|
+
stepsTaken: results.length,
|
|
428
|
+
totalSteps: steps.length,
|
|
429
|
+
minSteps,
|
|
430
|
+
imagesSent: frames.length,
|
|
431
|
+
escalations,
|
|
432
|
+
verdicts,
|
|
433
|
+
completed: !failed && results.length === steps.length,
|
|
434
|
+
}));
|
|
435
|
+
} catch {
|
|
436
|
+
/* as above: a flow that ran is not a flow that failed because of a log */
|
|
437
|
+
}
|
|
438
|
+
|
|
260
439
|
return {
|
|
261
440
|
device,
|
|
262
441
|
// Returned so a run that verified end to end can be handed straight to
|
|
263
442
|
// navigate.saveFlow without the caller reassembling what it just ran.
|
|
264
443
|
steps,
|
|
444
|
+
flowId,
|
|
265
445
|
endScreen,
|
|
266
446
|
results,
|
|
267
447
|
ok: !failed,
|
|
268
|
-
totalMs:
|
|
448
|
+
totalMs: wallMs,
|
|
269
449
|
ranSteps: results.length,
|
|
270
450
|
totalSteps: steps.length,
|
|
271
451
|
frames,
|