simframe 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,254 @@
1
+ #!/usr/bin/env node
2
+ // The agent half of the Human Parity Index.
3
+ //
4
+ // Runs every flow in the suite N times, exactly as an agent would run it, and
5
+ // reduces the flow records to an HPI report. The human half is recorded once
6
+ // by a person (`simframe baseline record`) and committed; this side is
7
+ // re-measured on every run, which is what makes HPI a trend rather than a
8
+ // claim.
9
+ //
10
+ // Two rules this file follows, both learned in this repo:
11
+ //
12
+ // - Nothing is verified through a pipe. `node script.mjs | tail -3` exits
13
+ // with tail's status, and this project has already shipped a check that
14
+ // could not fail because of it. Every gate here is an exit code.
15
+ // - A missing human baseline is reported, never defaulted. HPI_time with no
16
+ // denominator is null; it is not 1.0, and it is not "parity".
17
+ import fs from 'node:fs';
18
+ import path from 'node:path';
19
+ import { fileURLToPath } from 'node:url';
20
+ import { runScript } from '../src/actions.js';
21
+ import * as api from '../src/index.js';
22
+ import * as baseline from '../src/baseline.js';
23
+ import * as metrics from '../src/metrics.js';
24
+
25
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
26
+ const arg = (name, fallback = null) => {
27
+ const hit = process.argv.find((a) => a.startsWith(`--${name}=`));
28
+ return hit ? hit.slice(name.length + 3) : fallback;
29
+ };
30
+ const has = (name) => process.argv.includes(`--${name}`);
31
+
32
+ const device = arg('device');
33
+ const runs = Math.max(1, Number(arg('runs', '3')));
34
+ const only = arg('flow');
35
+ const out = arg('out');
36
+ const baselineFile = arg('baseline', path.join(ROOT, 'docs', 'research', 'hpi-baseline.json'));
37
+ const gate = has('gate');
38
+ /**
39
+ * How many times to measure the whole suite.
40
+ *
41
+ * Three when gating, because one measurement is mostly noise: identical code
42
+ * measured three times the same afternoon gave HPI_time 0.475, 0.413 and
43
+ * 0.371. The gate reads the median of the passes, which is what lets the time
44
+ * band stay as tight as 25% instead of being widened to cover a single run's
45
+ * spread.
46
+ */
47
+ const passes = Math.max(1, Number(arg('passes', gate ? '3' : '1')));
48
+ /**
49
+ * A pause between runs, and it is not superstition.
50
+ *
51
+ * Measured on one 8-run loop: 15.2 s, 15.2 s, then 6.9 s failing, then 25.7 s,
52
+ * 25.7 s, then the display wedged — and it wedged again within a couple of
53
+ * minutes of the same load after a device restart. Relaunching an app as fast
54
+ * as a script can is not what this suite is trying to measure, and a simulator
55
+ * asked to do it degrades and then stops rendering. The suite paces itself so
56
+ * the numbers describe simframe rather than the simulator's tolerance for
57
+ * being hammered.
58
+ */
59
+ const cooldownMs = Math.max(0, Number(arg('cooldown', '1500')));
60
+
61
+ const suite = baseline.loadSuite(arg('suite', baseline.SUITE_FILE)).filter((f) => !only || f.name === only);
62
+ if (!suite.length) {
63
+ console.error(`no flows to run${only ? ` matching "${only}"` : ''}`);
64
+ process.exit(2);
65
+ }
66
+
67
+ const { device: dev } = await api.ensureDaemon(device);
68
+ console.log(`device: ${dev.name} (${dev.udid})`);
69
+ console.log(`suite: ${suite.map((f) => f.name).join(', ')} × ${runs} run(s) × ${passes} pass(es)\n`);
70
+
71
+ const ids = new Set();
72
+ let hardFailures = 0;
73
+ /**
74
+ * A wedged capture loop is not a slow flow, and every run after it is doomed.
75
+ *
76
+ * Measured four times in one session: the display surface stops being readable
77
+ * mid-suite, the daemon re-resolves the display port and still gets nothing,
78
+ * and only restarting the device cures it. Grinding through the remaining runs
79
+ * produced three identical "could not run at all" lines and an HPI computed
80
+ * from whatever happened to finish first — a number with a hole in it, which
81
+ * is worse than no number.
82
+ */
83
+ const WEDGED = /display surface could not be read|did not produce a frame/;
84
+ /**
85
+ * The host could not do the thing, as distinct from simframe doing it slowly.
86
+ *
87
+ * A hosted runner takes 47-55 s to fail `simctl launch com.apple.Preferences`
88
+ * and then fails it again, three runs in a row — the same class of fault this
89
+ * repo already records for `simctl openurl`, which returns "Operation timed
90
+ * out" on a loaded runner. Six of those is nine minutes of CI spent measuring
91
+ * the runner's patience, and the resulting HPI describes nothing.
92
+ */
93
+ const ENVIRONMENT = /could not launch|Command failed: xcrun simctl (launch|terminate)|Operation timed out|timed out/i;
94
+ /** Two environmental failures of the same flow is the environment, not a flake. */
95
+ const ENV_GIVE_UP = 2;
96
+ let abort = null;
97
+ const envFailures = new Map();
98
+
99
+ const passSets = [];
100
+ outer: for (let pass = 1; pass <= passes; pass += 1) {
101
+ const thisPass = new Set();
102
+ passSets.push(thisPass);
103
+ if (passes > 1) console.log(`pass ${pass}/${passes}`);
104
+ for (const flow of suite) {
105
+ for (let run = 1; run <= runs; run += 1) {
106
+ // The same start state the human baseline was recorded from: app not
107
+ // running, on the home screen. Not part of the timed flow, and
108
+ // deliberately not expressed as flow steps — see baseline.resetFor.
109
+ // The reset and its settle live inside the same guard as the run.
110
+ //
111
+ // They did not, and a device whose display had failed threw out of
112
+ // `waitFor` — outside the try — killing the process with a stack trace
113
+ // instead of the one classification this script exists to make. An
114
+ // environmental failure has to be classified wherever it happens, not
115
+ // only where it was convenient to catch.
116
+ let res;
117
+ try {
118
+ const reset = await baseline.resetFor(dev.udid, flow);
119
+ if (reset.failures.length) console.log(` (reset: ${reset.failures.join('; ')})`);
120
+ await api.waitFor(dev.udid, { mode: 'stable', stableMs: 400, timeoutMs: 4000 });
121
+ res = await runScript(dev.udid, {
122
+ steps: flow.steps,
123
+ flowName: flow.name,
124
+ minSteps: flow.minSteps ?? null,
125
+ });
126
+ } catch (err) {
127
+ // A flow that could not start at all is not a slow flow. It has no
128
+ // record, so it cannot be averaged into anything; it is reported and
129
+ // counted.
130
+ hardFailures += 1;
131
+ console.log(`FAIL ${flow.name} run ${run}: ${err.message}`);
132
+ if (WEDGED.test(err.message)) {
133
+ abort = { kind: 'capture is wedged', message: err.message };
134
+ break outer;
135
+ }
136
+ if (ENVIRONMENT.test(err.message)) {
137
+ const n = (envFailures.get(flow.name) ?? 0) + 1;
138
+ envFailures.set(flow.name, n);
139
+ if (n >= ENV_GIVE_UP) {
140
+ abort = { kind: `the host cannot run "${flow.name}"`, message: err.message.split('\n')[0] };
141
+ break outer;
142
+ }
143
+ }
144
+ continue;
145
+ }
146
+ ids.add(res.flowId);
147
+ thisPass.add(res.flowId);
148
+ const verdicts = res.results.map((r) => r.verification?.verdict).filter(Boolean);
149
+ console.log(
150
+ `${res.ok ? 'ok ' : 'FAIL'} ${flow.name.padEnd(24)} run ${run}/${runs} ` +
151
+ `${String(res.totalMs).padStart(6)}ms ${res.ranSteps}/${res.totalSteps} steps ` +
152
+ `${verdicts.filter((v) => v !== 'ok').length ? `verdicts: ${verdicts.join(',')}` : 'all ok'}`,
153
+ );
154
+ if (!res.ok) console.log(` ${res.results.filter((r) => !r.ok).map((r) => r.error).join('; ')}`);
155
+ if (cooldownMs) await new Promise((r) => setTimeout(r, cooldownMs));
156
+ }
157
+ }
158
+ }
159
+
160
+ // This process's runs only. The log is append-only and holds every earlier
161
+ // measurement, which is what makes it a trend — but a CI number computed over
162
+ // somebody's local runs from last week is not this commit's number.
163
+ const flows = metrics.readFlows(dev.udid).filter((f) => ids.has(f.flow_id));
164
+ const humans = baseline.readBaselines();
165
+ // Each pass measured on its own, so the gate can take a median over them
166
+ // rather than trusting one. Accuracy is pooled over every run instead: one
167
+ // wrong action in thirty is a wrong action, and a median would hide it.
168
+ const allFlows = metrics.readFlows(dev.udid);
169
+ const passReports = passSets
170
+ .map((set) => metrics.hpi({ flows: allFlows.filter((f) => set.has(f.flow_id)), baselines: humans }))
171
+ .filter((r) => r.overall.runs > 0);
172
+ const passTimes = passReports.map((r) => r.overall.hpi_time).filter((t) => Number.isFinite(t));
173
+
174
+ const report = {
175
+ ...metrics.hpi({ flows, baselines: humans }),
176
+ measured_at: new Date().toISOString(),
177
+ device: { udid: dev.udid, name: dev.name, runtime: dev.runtime },
178
+ runs_per_flow: runs,
179
+ passes,
180
+ pass_hpi_time: passTimes,
181
+ hard_failures: hardFailures,
182
+ human_baselines: Object.fromEntries(
183
+ Object.entries(humans).map(([k, v]) => [k, { runs: v.runs, p50: v.wall_time_ms?.p50 ?? null }]),
184
+ ),
185
+ };
186
+
187
+ console.log('\nflow runs agent p50 human p50 HPI_time step_ratio');
188
+ for (const f of report.flows) {
189
+ console.log(
190
+ `${f.flow.padEnd(24)} ${String(f.runs).padStart(5)} ${`${f.agent_ms.p50}ms`.padStart(9)} ` +
191
+ `${(f.human_median_ms ? `${f.human_median_ms}ms` : '—').padStart(9)} ` +
192
+ `${String(f.hpi_time ?? '—').padStart(8)} ${String(f.step_ratio ?? '—').padStart(10)}`,
193
+ );
194
+ }
195
+ report.overall.hpi_time_median_of_passes = passTimes.length ? Number(metrics.median(passTimes).toFixed(3)) : null;
196
+ const o = report.overall;
197
+ console.log(`\nHPI_accuracy ${o.hpi_accuracy} HPI_time ${o.hpi_time ?? '—'} HPI ${o.hpi ?? '—'} step_ratio ${o.step_ratio ?? '—'}`);
198
+ if (passTimes.length > 1) {
199
+ console.log(`HPI_time per pass: ${passTimes.join(', ')} — median ${o.hpi_time_median_of_passes} (what the gate reads)`);
200
+ }
201
+ if (o.hpi_time == null) {
202
+ console.log(`no human baseline for any measured flow — HPI_time and HPI are null, not 1.0.`);
203
+ console.log(`record one: simframe baseline record <flow> --device=${dev.udid} --runs=5`);
204
+ }
205
+
206
+ if (out) {
207
+ fs.writeFileSync(out, `${JSON.stringify(report, null, 2)}\n`);
208
+ console.log(`wrote ${out}`);
209
+ }
210
+
211
+ const escalations = metrics.breakdown(metrics.readEscalations(dev.udid));
212
+ console.log(`\nescalations in this device's log: ${escalations.total} (${Object.entries(escalations.by_reason).filter(([, n]) => n).map(([r, n]) => `${r} ${n}`).join(', ') || 'none'})`);
213
+
214
+ if (abort) {
215
+ console.error(`\n${abort.kind}: ${abort.message}`);
216
+ console.error('No comparable HPI was measured — what is above is a partial suite and');
217
+ console.error('must not be adopted as a baseline or read as a regression.');
218
+ if (/wedged/.test(abort.kind)) {
219
+ console.error('Only restarting the device is known to cure a wedge:');
220
+ console.error(` xcrun simctl shutdown ${dev.udid} && xcrun simctl boot ${dev.udid}`);
221
+ }
222
+ // 2, not 1. "I could not measure" and "it got worse" are different answers,
223
+ // and a job that reports them with the same exit code teaches people to
224
+ // ignore both. The workflow treats 2 as a loud warning and 1 as a failure.
225
+ process.exit(2);
226
+ }
227
+
228
+ if (hardFailures) {
229
+ console.error(`\n${hardFailures} flow run(s) could not run at all.`);
230
+ process.exit(1);
231
+ }
232
+
233
+ if (!gate) process.exit(0);
234
+
235
+ // ------------------------------------------------------------------ the gate
236
+ const committed = fs.existsSync(baselineFile) ? JSON.parse(fs.readFileSync(baselineFile, 'utf8')) : null;
237
+ if (!committed) {
238
+ // Loudly inactive rather than quietly passing. A gate with nothing to
239
+ // compare against cannot detect a regression, and saying "ok" here is how a
240
+ // CI job comes to mean nothing.
241
+ console.log(`\nGATE INACTIVE — no committed baseline at ${path.relative(ROOT, baselineFile)}.`);
242
+ console.log('Commit this run as the baseline to arm it.');
243
+ process.exit(0);
244
+ }
245
+
246
+ const base = committed.overall ?? {};
247
+ const failures = metrics.gateAgainst(committed, report);
248
+
249
+ console.log(`\ngate vs ${path.relative(ROOT, baselineFile)} (measured ${committed.measured_at ?? '?'})`);
250
+ console.log(` HPI_accuracy ${base.hpi_accuracy ?? '—'} -> ${o.hpi_accuracy ?? '—'}`);
251
+ console.log(` HPI_time ${metrics.gateTime(base) ?? '—'} -> ${metrics.gateTime(o) ?? '—'}`);
252
+ console.log(` band ${metrics.TIME_REGRESSION * 100}% (median of ${passTimes.length || 1} pass(es))`);
253
+ for (const f of failures) console.log(`FAIL ${f}`);
254
+ process.exit(failures.length ? 1 : 0);
@@ -75,6 +75,13 @@ function requiredFiles() {
75
75
  throw new Error('no skill found under skills/ — has the layout moved? This check would silently pass.');
76
76
  }
77
77
 
78
+ // The flow suite. `simframe baseline` and `simframe hpi` read it at
79
+ // runtime, so an absent one is a command that only fails once installed.
80
+ for (const f of walk(path.join(ROOT, 'flows'), (p) => p.endsWith('.json'))) required.add(rel(f));
81
+ if (!walk(path.join(ROOT, 'flows'), (p) => p.endsWith('.json')).length) {
82
+ throw new Error('no flow suite found under flows/ — has the layout moved? This check would silently pass.');
83
+ }
84
+
78
85
  // Anything package.json points at to run.
79
86
  const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8'));
80
87
  for (const target of Object.values(pkg.bin ?? {})) required.add(target.replace(/^\.\//, ''));
package/src/actions.js CHANGED
@@ -6,6 +6,8 @@ import * as api from './index.js';
6
6
  import * as graph from './graph.js';
7
7
  import * as input from './input.js';
8
8
  import * as intent from './intent.js';
9
+ import * as metrics from './metrics.js';
10
+ import * as screenmap from './screenmap.js';
9
11
  import { launchApp, openUrl, setPermission, terminateApp } from './platform/index.js';
10
12
 
11
13
  const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
@@ -112,6 +114,12 @@ export async function runScript(
112
114
  // Rebuild the HID session and retry once when a hardware button provably
113
115
  // did nothing. Off only for a caller deliberately testing that path.
114
116
  recoverInput = true,
117
+ // What this run is called and how few steps it could take, for the flow
118
+ // record. A bare `sim_do` has neither and says so with nulls rather than
119
+ // inventing a name — an unnamed run still gets timed, it just cannot be
120
+ // compared against a human baseline.
121
+ flowName = null,
122
+ minSteps = null,
115
123
  options,
116
124
  } = {},
117
125
  ) {
@@ -119,6 +127,24 @@ export async function runScript(
119
127
  const { device } = await api.ensureDaemon(deviceQuery, options);
120
128
  const udid = device.udid;
121
129
  const startedAt = Date.now();
130
+ // Measurement only. Nothing below reads these, and a failure to write one
131
+ // can never change what a step does — see `note`.
132
+ const flowId = metrics.newFlowId();
133
+ const escalations = [];
134
+ const verdicts = [];
135
+ // Named noteEscalation, not note: the step loop below declares its own
136
+ // `note` string for the no-visible-change suffix, which shadowed this and
137
+ // turned every escalating verdict into a failed step reading "note is not a
138
+ // function". The try/catch inside here could not help — the throw was at the
139
+ // call site, one scope out. Instrumentation that can fail a flow is worse
140
+ // than no instrumentation.
141
+ const noteEscalation = (record) => {
142
+ try {
143
+ escalations.push(metrics.recordEscalation(udid, { flowId, ...record }));
144
+ } catch {
145
+ /* instrumentation must not be able to fail a flow it is only watching */
146
+ }
147
+ };
122
148
 
123
149
  const needsInput = steps.some((s) => ACTION_STEPS.has(normalizeStep(s).action));
124
150
  if (needsInput) {
@@ -165,13 +191,58 @@ export async function runScript(
165
191
  const prediction = verify && beforeScreen?.hash ? graph.predict(udid, beforeScreen, step) : null;
166
192
  try {
167
193
  let detail = await runStep(deviceQuery, udid, step, { screen, options, frames });
194
+ // How long this transition has cost before, on this screen, for this
195
+ // action. A cold edge gets the old fixed default and says so; a measured
196
+ // one gets p95 plus a margin. Research §7.
197
+ const learned = verify && beforeScreen?.hash ? graph.timingFor(udid, beforeScreen, step) : null;
198
+ // How long this screen must hold still before it counts as settled.
199
+ //
200
+ // 500 ms was a constant paid by every step of every flow, and it is the
201
+ // reducible half of a settle: the rest is the transition genuinely
202
+ // taking time. An edge whose transition has never paused mid-flight
203
+ // needs 150 ms of quiet, not 500. Capped at the caller's own value, so
204
+ // this can only ever shorten a wait.
205
+ // NOT YET USED TO DECIDE ANYTHING, and the reason is worth the space.
206
+ //
207
+ // `graph.stillnessFor` computes a shorter window from the longest pause
208
+ // ever observed inside this transition, and measured live it made the
209
+ // Settings flow 8.0 s instead of 11.5 s — and wrong. Eight runs in a row
210
+ // failed at step 2 with the screen still showing Settings root, because
211
+ // step 1's settle returned mid-push, `screenIdentity` then read the
212
+ // screen we had not left yet, and the graph learned root -> root as a
213
+ // verified edge and started predicting it.
214
+ //
215
+ // The flaw is in the estimator, not the idea: the gap statistic is
216
+ // gathered only from what a wait itself observed, so a wait that ends
217
+ // early never sees the pauses that come later, the gaps look like zero,
218
+ // the window ratchets down, and the next wait ends earlier still. A
219
+ // self-reinforcing bias with a wrong graph at the end of it.
220
+ //
221
+ // The unbiased estimator is available and is a separate piece of work:
222
+ // the frame history holds every frame's timestamp and diff, so the true
223
+ // motion profile of a transition can be computed *after* it is over
224
+ // rather than from inside the wait that cut it short. Until then the
225
+ // gaps are recorded and not acted on — measuring is safe, and this is
226
+ // Phase 11's own rule that a learned number may only ever shorten a
227
+ // wait, applied to itself.
228
+ const stillness = step.stableMs ?? stableMs;
229
+ const stillnessPlan = learned ? graph.stillnessFor(learned, stillness) : { cold: true };
230
+ // A settle is not satisfied until the screen has held still for
231
+ // `stillness`, so a budget below that can never be met — and the learned
232
+ // p95 is measured from waits that include the stillness window, which
233
+ // makes it self-consistent but not self-evidently so. A tab switch with
234
+ // a p95 of 90ms would get a 240ms budget and then time out at 240ms
235
+ // waiting for 500ms of quiet, turning every fast edge into a failure.
236
+ const floorMs = stillness + 250;
237
+ const budgetMs = step.timeoutMs
238
+ ?? (learned && !learned.cold ? Math.max(learned.timeoutMs, floorMs) : timeoutMs);
168
239
  const settleFor = async () => {
169
240
  if (!autoSettle || !ACTION_STEPS.has(step.action)) return null;
170
241
  const w = await api.waitFor(deviceQuery, {
171
242
  mode: 'settle',
172
243
  since: before,
173
- stableMs: step.stableMs ?? stableMs,
174
- timeoutMs: step.timeoutMs ?? timeoutMs,
244
+ stableMs: stillness,
245
+ timeoutMs: budgetMs,
175
246
  options,
176
247
  });
177
248
  return {
@@ -180,9 +251,55 @@ export async function runScript(
180
251
  sawChange: w.sawChange,
181
252
  stalled: Boolean(w.stalled),
182
253
  noVisibleChange: Boolean(w.noVisibleChange),
254
+ // What this wait was allowed, and where the number came from. A
255
+ // timeout nobody can explain is how a fixed sleep comes back as a
256
+ // constant with a comment.
257
+ budgetMs,
258
+ stillnessMs: stillness,
259
+ quietGapMs: w.quietGapMs,
260
+ timing: learned
261
+ ? {
262
+ p50: learned.p50,
263
+ p95: learned.p95,
264
+ samples: learned.samples,
265
+ cold: learned.cold,
266
+ gapP95: learned.gapP95,
267
+ gapSamples: learned.gapSamples,
268
+ // What it *would* have been, for the eval that has to happen
269
+ // before this is trusted with a wait.
270
+ stillnessWouldBe: stillnessPlan.stillnessMs ?? null,
271
+ }
272
+ : null,
183
273
  };
184
274
  };
185
275
  let settled = await settleFor();
276
+ // A screen that is still working earns more time; a screen doing nothing
277
+ // visible has already answered. Research §7: keep waiting past p95 only
278
+ // while the transition classifier says something is loading, and never
279
+ // past Nielsen's 10 s — at which point it escalates with the timing
280
+ // attached rather than waiting longer.
281
+ if (settled && !settled.ok && !settled.noVisibleChange && learned && !learned.cold) {
282
+ const kind = (await api.getState(deviceQuery, { options })).state.transition?.kind;
283
+ const verdict = graph.slowerThanUsual({
284
+ elapsedMs: settled.waitedMs, p95: learned.p95, settled: false, kind,
285
+ });
286
+ if (verdict.keepWaiting) {
287
+ const remaining = graph.HARD_CAP_MS - settled.waitedMs;
288
+ const more = await api.waitFor(deviceQuery, {
289
+ mode: 'settle', since: before, stableMs: stillness, timeoutMs: remaining, options,
290
+ });
291
+ settled = {
292
+ ...settled,
293
+ ok: more.satisfied,
294
+ waitedMs: settled.waitedMs + more.waitedMs,
295
+ sawChange: settled.sawChange || more.sawChange,
296
+ quietGapMs: Math.max(settled.quietGapMs ?? 0, more.quietGapMs ?? 0),
297
+ slowerThanUsual: verdict.note,
298
+ };
299
+ } else if (verdict.slower) {
300
+ settled = { ...settled, slowerThanUsual: verdict.note };
301
+ }
302
+ }
186
303
 
187
304
  // A hardware button that moved nothing did not arrive.
188
305
  //
@@ -227,7 +344,19 @@ export async function runScript(
227
344
  // nothing at all.
228
345
  endScreen = afterScreen;
229
346
  if (afterScreen.confirmed && afterScreen.hash) {
230
- graph.record(udid, { from: beforeScreen, action: step, to: afterScreen, kind });
347
+ // The observed cost of this transition, which is what makes the next
348
+ // one adaptive. Only from a settle that was actually satisfied: a
349
+ // timeout is not a measurement of how long the screen takes, it is a
350
+ // measurement of how long we were prepared to wait.
351
+ graph.record(udid, {
352
+ from: beforeScreen, action: step, to: afterScreen, kind,
353
+ settleMs: settled?.ok ? settled.waitedMs : undefined,
354
+ // The pause statistic is worth having from any settle that saw the
355
+ // screen move, satisfied or not: a transition that paused for
356
+ // 400ms and then timed out is exactly the case a 150ms stillness
357
+ // window would have got wrong.
358
+ quietGapMs: settled?.sawChange ? settled.quietGapMs : undefined,
359
+ });
231
360
  carriedScreen = afterScreen;
232
361
  }
233
362
  }
@@ -244,6 +373,25 @@ export async function runScript(
244
373
  settled,
245
374
  });
246
375
  const halt = haltDecision({ verification, stopOnUnexpected, continueOnError });
376
+ if (verification?.verdict) verdicts.push(verification.verdict);
377
+ if (metrics.ESCALATING_VERDICTS.has(verification?.verdict)) {
378
+ noteEscalation({
379
+ stepIndex: i,
380
+ fingerprint: beforeScreen?.hash ?? null,
381
+ reason: 'verification_failed',
382
+ candidates: [],
383
+ // A halted run is a decision simframe made and stopped on; a step
384
+ // that moved nothing carries on and leaves the judgement to whoever
385
+ // reads the result.
386
+ outcome: halt.halt ? 'failed' : 'escalated_to_model',
387
+ wallMs: Date.now() - stepStart,
388
+ detail: `${verification.verdict}: ${verification.detail}`
389
+ + (settled?.slowerThanUsual ? ` [${settled.slowerThanUsual}]` : '')
390
+ + (settled?.timing && !settled.timing.cold
391
+ ? ` [waited ${settled.waitedMs}ms of a ${settled.budgetMs}ms budget; p95 ${settled.timing.p95}ms]`
392
+ : ''),
393
+ });
394
+ }
247
395
  if (halt.halt) {
248
396
  results[results.length - 1].ok = false;
249
397
  results[results.length - 1].error = halt.error;
@@ -252,20 +400,52 @@ export async function runScript(
252
400
  }
253
401
  } catch (err) {
254
402
  results.push({ index: i, action: step.action, ok: false, ms: Date.now() - stepStart, error: err.message });
403
+ const why = metrics.reasonForStepError(step, err);
404
+ noteEscalation({
405
+ stepIndex: i,
406
+ fingerprint: beforeScreen?.hash ?? metrics.fingerprintNow(udid, screenmap),
407
+ reason: why.reason,
408
+ candidates: why.candidates,
409
+ tried: why.tried,
410
+ outcome: 'failed',
411
+ wallMs: Date.now() - stepStart,
412
+ detail: err.message,
413
+ });
255
414
  failed = true;
256
415
  if (!continueOnError) break;
257
416
  }
258
417
  }
259
418
 
419
+ const wallMs = Date.now() - startedAt;
420
+ try {
421
+ metrics.recordFlow(udid, metrics.flowRecordFrom({
422
+ flowId,
423
+ flowName,
424
+ udid,
425
+ startedAt,
426
+ wallMs,
427
+ stepsTaken: results.length,
428
+ totalSteps: steps.length,
429
+ minSteps,
430
+ imagesSent: frames.length,
431
+ escalations,
432
+ verdicts,
433
+ completed: !failed && results.length === steps.length,
434
+ }));
435
+ } catch {
436
+ /* as above: a flow that ran is not a flow that failed because of a log */
437
+ }
438
+
260
439
  return {
261
440
  device,
262
441
  // Returned so a run that verified end to end can be handed straight to
263
442
  // navigate.saveFlow without the caller reassembling what it just ran.
264
443
  steps,
444
+ flowId,
265
445
  endScreen,
266
446
  results,
267
447
  ok: !failed,
268
- totalMs: Date.now() - startedAt,
448
+ totalMs: wallMs,
269
449
  ranSteps: results.length,
270
450
  totalSteps: steps.length,
271
451
  frames,