simframe 0.14.2 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,171 @@
1
+ // Did the app we launched actually come forward?
2
+ //
3
+ // `launch` could not answer that, and item 169 is what it cost: `launch
4
+ // com.apple.Preferences (relaunch: true)` returning ok, reporting `[no visible
5
+ // change]`, with the device still on the previous app. The CI workflow's own
6
+ // step vehicle retries `simctl launch` three times by hand for exactly this, so
7
+ // the failure was known at the harness layer and unhandled at the library
8
+ // layer, where every user meets it.
9
+ //
10
+ // **Screen change cannot settle it, and that dead end was checked first.**
11
+ // Relaunching an app that is already frontmost legitimately lands on the same
12
+ // screen, so "no visible change" is shared by the success and the failure. The
13
+ // discriminator has to be identity.
14
+ //
15
+ // The identity is a **pid**, not a name, and that is the whole reason this works
16
+ // cheaply: `simctl launch` prints the pid it started, and the daemon's
17
+ // `frontmost` action reports the pid of the application AXPTranslator says is in
18
+ // front. Measured on a real device — 10695/10695 for Preferences,
19
+ // 10762/10762 for Contacts — so the caller compares two integers instead of
20
+ // matching a display name against a bundle id.
21
+ //
22
+ // Its own module, with the device read injectable, because the only thing that
23
+ // ever exercises a launch is a device: a unit test replays pid sequences
24
+ // through `landed` and never boots anything.
25
+ import * as control from './control.js';
26
+
27
+ /**
28
+ * How long to wait for the app to reach the front.
29
+ *
30
+ * Measured, not chosen for feel: an already-running app fronts in ~240 ms and a
31
+ * cold switch to Contacts took 1552 ms on this machine. 5 s leaves room for a
32
+ * loaded runner — the same host where `simctl launch` itself has measured
33
+ * 47–55 s — while still being far below the point where a caller gives up.
34
+ */
35
+ export const FRONT_BUDGET_MS = 5000;
36
+ export const POLL_MS = 100;
37
+
38
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
39
+
40
+ /**
41
+ * Words the translator hands back when it has no name to give.
42
+ *
43
+ * Measured on a device: the application element's title is "Settings" while
44
+ * Settings is on screen, and after a press of home the **same pid** is still
45
+ * frontmost with the title degraded to the bare word "application". That is
46
+ * not a name, and printing it as one would be a confident wrong answer in
47
+ * precisely the state worth noticing — an app frontmost by pid that has
48
+ * stopped naming itself. So the word is kept (the daemon reports raw) and
49
+ * phrased honestly here, where it can be tested without a device.
50
+ */
51
+ const GENERIC_NAMES = new Set(['application', 'window', 'unknown', 'group', 'element']);
52
+
53
+ /** How to refer to whoever holds the front. */
54
+ export function nameHolder(pid, title) {
55
+ if (pid === null || pid === undefined) return 'nothing';
56
+ // Blank counts as absent, not as a degraded answer: whitespace is the
57
+ // translator saying nothing, and "no longer names itself" is a claim about
58
+ // an app that answered.
59
+ const said = typeof title === 'string' ? title.trim() : '';
60
+ if (!said) return `pid ${pid}`;
61
+ return GENERIC_NAMES.has(said.toLowerCase())
62
+ ? `pid ${pid} (an app that no longer names itself)`
63
+ : `pid ${pid} (${said})`;
64
+ }
65
+
66
+ /** Who is on screen — pid to compare, title to report. Nulls mean "cannot say". */
67
+ export async function read(udid) {
68
+ if (!udid || !control.available(udid)) return { pid: null, title: null };
69
+ try {
70
+ const r = await control.request(udid, { action: 'frontmost' });
71
+ return { pid: typeof r?.pid === 'number' ? r.pid : null, title: r?.title ?? null };
72
+ } catch {
73
+ // A daemon that cannot answer is "cannot say". Reporting it as "not that
74
+ // app" would turn a missing sensor into a failed launch, which is the
75
+ // false refusal item 161 was reverted for.
76
+ return { pid: null, title: null };
77
+ }
78
+ }
79
+
80
+ /** The pid alone, for callers that only compare. */
81
+ export const frontmostPid = async (udid) => (await read(udid)).pid;
82
+
83
+ /**
84
+ * Wait for `pid` to be the app in front.
85
+ *
86
+ * Three verdicts, and the third is not a failure:
87
+ *
88
+ * - `fronted` — the launched pid is the frontmost pid. The launch worked,
89
+ * whether or not the screen moved.
90
+ * - `did-not-front` — the budget expired with someone else in front. This is
91
+ * the defect, now visible.
92
+ * - `cannot-say` — no pid from the launch, or nothing on this platform reports
93
+ * who is frontmost. The caller keeps whatever it said before this existed.
94
+ *
95
+ * @param {object} o
96
+ * @param {number|null} o.pid what the launch reported
97
+ * @param {() => Promise<{pid: number|null, title: string|null}>} o.read who is in front now
98
+ */
99
+ export async function landed({
100
+ pid, read, budgetMs = FRONT_BUDGET_MS, pollMs = POLL_MS,
101
+ now = () => Date.now(), wait = sleep,
102
+ } = {}) {
103
+ if (typeof pid !== 'number') {
104
+ return { verdict: 'cannot-say', reason: 'the launch did not report a pid' };
105
+ }
106
+ const started = now();
107
+ let seen = null;
108
+ let asked = 0;
109
+ // Who held the front while we waited, in order. The instrument, not decoration:
110
+ // "the budget was too short" and "the app never went anywhere" produce the same
111
+ // verdict and want opposite remedies, and one list of pids tells them apart —
112
+ // a front that changed hands twice is a slow device, a single pid for the whole
113
+ // budget is a launch that did not happen. Widening a budget without this is the
114
+ // mistake item 146 was, twice.
115
+ const held = [];
116
+ let holder = null;
117
+ for (;;) {
118
+ const look = await read();
119
+ seen = look?.pid ?? null;
120
+ asked += 1;
121
+ if (held[held.length - 1] !== seen) held.push(seen);
122
+ if (seen !== null) holder = look;
123
+ if (seen === pid) return { verdict: 'fronted', ms: now() - started, polls: asked, held };
124
+ // Checked after at least one read, so a platform that cannot answer says so
125
+ // rather than spending the whole budget finding that out.
126
+ if (seen === null && asked === 1) {
127
+ return { verdict: 'cannot-say', reason: 'nothing on this device reports which app is frontmost' };
128
+ }
129
+ if (now() - started >= budgetMs) {
130
+ return {
131
+ verdict: 'did-not-front',
132
+ ms: now() - started,
133
+ polls: asked,
134
+ frontmost: seen,
135
+ held,
136
+ // Named, not just numbered. A runner held the front at pid 7797 through
137
+ // nine consecutive failed launches and the only question that mattered
138
+ // — *what* is 7797 — was the one a number could not answer.
139
+ holder: nameHolder(seen, holder?.pid === seen ? holder.title : null),
140
+ };
141
+ }
142
+ await wait(pollMs);
143
+ }
144
+ }
145
+
146
+ /**
147
+ * `held` as a sentence, because a list of pids is not a diagnosis by itself.
148
+ *
149
+ * **Only meaningful for a `did-not-front`.** Handed a successful wait's `held`
150
+ * it says the launch never took effect about a launch that plainly did — I did
151
+ * exactly that while testing this — so `pid` is taken and the success is
152
+ * refused rather than described. Callers that hold the verdict gate on it;
153
+ * this is the belt for the one that forgets.
154
+ */
155
+ export function describeHeld(held = [], pid = null) {
156
+ const real = held.filter((p) => p !== null);
157
+ if (pid !== null && real[real.length - 1] === pid) {
158
+ return `pid ${pid} did reach the front — there is nothing to explain`;
159
+ }
160
+ if (real.length <= 1) {
161
+ return real.length === 1
162
+ ? `pid ${real[0]} held the front for the whole wait, so the launch never took effect`
163
+ : 'nothing held the front at any point';
164
+ }
165
+ return `the front changed hands ${real.length - 1} time(s) (${real.join(' → ')}),`
166
+ + ' so the device was switching apps and simply never reached this one';
167
+ }
168
+
169
+ /** `landed`, reading from the daemon. */
170
+ export const check = (udid, pid, opts = {}) =>
171
+ landed({ pid, read: () => read(udid), ...opts });
package/src/metrics.js CHANGED
@@ -610,13 +610,31 @@ export function hpi({ flows, baselines = {} }) {
610
610
  }
611
611
 
612
612
  const perFlow = [...byName.entries()].map(([name, runs]) => {
613
- const agent = quartiles(runs.map((r) => r.wall_time_ms));
613
+ // **Time is measured over runs that finished the flow.** This took every
614
+ // run's wall time, failures included, and a failure is fast — so a change
615
+ // that broke a flow registered as the agent getting quicker.
616
+ //
617
+ // Measured, not hypothesised: `settings-larger-text` failed all three runs
618
+ // at ~3.6 s against a 7799 ms human median and reported `hpi_time 2.163`,
619
+ // i.e. "twice as fast as a person", about a flow that never once reached
620
+ // its destination. The composite `hpi` survives that because accuracy
621
+ // divides it down — but the CI gate's threshold is written against
622
+ // `HPI_time`, so the one number the gate reads was the one being flattered
623
+ // by breakage.
624
+ //
625
+ // A flow where nothing completed reports no time at all rather than a
626
+ // flattering one. That is the same discipline as the suite's own "no
627
+ // comparable HPI was measured": a number that cannot be compared must not
628
+ // be offered as one.
629
+ const finished = runs.filter((r) => r.completed);
630
+ const agent = quartiles(finished.map((r) => r.wall_time_ms));
614
631
  const human = baselines[name]?.wall_time_ms ?? null;
615
632
  const humanMedian = human?.p50 ?? null;
616
633
  const stepRatios = runs.map((r) => r.step_ratio).filter((x) => Number.isFinite(x));
617
634
  return {
618
635
  flow: name,
619
636
  runs: runs.length,
637
+ timed_runs: finished.length,
620
638
  agent_ms: agent,
621
639
  human_median_ms: humanMedian,
622
640
  hpi_time: humanMedian && agent?.p50 ? Number((humanMedian / agent.p50).toFixed(3)) : null,
@@ -993,6 +993,16 @@ function capabilities() {
993
993
  supported: false,
994
994
  note: 'not built for Android yet — `uiautomator dump` costs ~2s a read; see docs/DEFERRED.md',
995
995
  },
996
+ // Not borrowed from iOS, and not claimed. iOS reads the frontmost pid off
997
+ // AXPTranslator and compares it with the pid `simctl launch` printed;
998
+ // neither half exists here — the emulator console launches by intent and
999
+ // reports no pid. `am start` does front the activity synchronously, which
1000
+ // is why this has not been the same problem, but "has not been" is not a
1001
+ // check and must not report as one.
1002
+ frontmost: {
1003
+ supported: false,
1004
+ note: 'no frontmost-app read on Android yet, so a launch is not confirmed to have fronted; `am start` fronts synchronously, which is why this has not bitten — see docs/DEFERRED.md',
1005
+ },
996
1006
  };
997
1007
  }
998
1008
 
@@ -264,6 +264,13 @@ async function screenshot(udid, outFile, { mask = 'ignored' } = {}) {
264
264
  * simctl passes launch arguments after the bundle id and environment through
265
265
  * `SIMCTL_CHILD_`-prefixed variables of its own process — which is why env has
266
266
  * to be set on the child rather than passed as flags.
267
+ *
268
+ * Returns the pid simctl started, because starting an app and *fronting* an app
269
+ * are different events and only the pid connects them: the layer above compares
270
+ * it against the pid the device says is in front (item 169). simctl prints
271
+ * `com.apple.Preferences: 10695` and has always printed it; nobody had read it.
272
+ * A shape we do not recognise gives null, which reads as "cannot say" upstairs
273
+ * and never as a failed launch.
267
274
  */
268
275
  async function launchApp(udid, bundleId, { args = [], env = {}, terminateFirst = false } = {}) {
269
276
  if (terminateFirst) {
@@ -279,10 +286,11 @@ async function launchApp(udid, bundleId, { args = [], env = {}, terminateFirst =
279
286
  const childEnv = { ...process.env };
280
287
  for (const [k, v] of Object.entries(env)) childEnv[`SIMCTL_CHILD_${k}`] = String(v);
281
288
  try {
282
- await run('xcrun', ['simctl', 'launch', udid, bundleId, ...args.map(String)], {
289
+ const { stdout } = await run('xcrun', ['simctl', 'launch', udid, bundleId, ...args.map(String)], {
283
290
  timeout: SIMCTL_TIMEOUT_MS,
284
291
  env: childEnv,
285
292
  });
293
+ return { pid: launchedPid(stdout) };
286
294
  } catch (err) {
287
295
  // execFile's message is just "Command failed: ..." with simctl's actual
288
296
  // complaint left in stderr. A CI run failed here and said nothing about
@@ -291,6 +299,12 @@ async function launchApp(udid, bundleId, { args = [], env = {}, terminateFirst =
291
299
  }
292
300
  }
293
301
 
302
+ /** The pid out of simctl's `<bundle-id>: <pid>`, or null if it said otherwise. */
303
+ export function launchedPid(stdout) {
304
+ const m = /:\s*(\d+)\s*$/.exec(String(stdout ?? '').trim());
305
+ return m ? Number(m[1]) : null;
306
+ }
307
+
294
308
  async function terminateApp(udid, bundleId) {
295
309
  try {
296
310
  await run('xcrun', ['simctl', 'terminate', udid, bundleId], { timeout: SIMCTL_TIMEOUT_MS });
@@ -567,6 +581,11 @@ function capabilities() {
567
581
  captureEngines: ['simframed', 'screenshot'],
568
582
  input: { supported: true, via: 'daemon' },
569
583
  ax: { supported: true },
584
+ // Whether a launch can be confirmed to have reached the front. Declared
585
+ // separately from `ax` even though the same bridge answers both, because
586
+ // the user-visible consequence is different: without it `launch` reports
587
+ // that a process started and calls that success (item 169).
588
+ frontmost: { supported: true, via: 'AXPTranslator, compared by pid' },
570
589
  };
571
590
  }
572
591