simframe 0.7.2 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/metrics.js ADDED
@@ -0,0 +1,499 @@
1
+ // Phase 10: simframe measuring itself.
2
+ //
3
+ // Two logs, both JSONL, both per device. `escalations.jsonl` records every
4
+ // point where simframe handed a decision back to the agent; `flows.jsonl`
5
+ // records one line per flow run. Nothing here changes what simframe does — it
6
+ // only writes down what happened, which is the only way the phases after this
7
+ // one can be prioritised or proven.
8
+ //
9
+ // The escalation log is the steering wheel: the reason breakdown decides which
10
+ // faculty is built next. So a reason is mandatory, "unknown" is not one of
11
+ // them, and every existing hand-back path maps to exactly one of the five.
12
+ import fs from 'node:fs';
13
+ import path from 'node:path';
14
+ import * as store from './store.js';
15
+
16
+ /** The five reasons, from docs/research/03-human-parity.md §8. Nothing else is a reason. */
17
+ export const REASONS = [
18
+ 'unknown_screen',
19
+ 'ambiguous_intent',
20
+ 'verification_failed',
21
+ 'novel_dialog',
22
+ 'no_plan',
23
+ ];
24
+
25
+ export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
26
+
27
+ /**
28
+ * Which faculty would have removed this escalation.
29
+ *
30
+ * This is the mapping that turns the reason breakdown into a phase order, so
31
+ * it lives next to the reasons rather than in prose. `built` is empty today
32
+ * and each phase moves its own faculty into it — see avoidableRate() for why
33
+ * that matters and what the rate honestly means before then.
34
+ */
35
+ export const FACULTY = {
36
+ unknown_screen: 'exploration (Phase 14)',
37
+ ambiguous_intent: 'icon semantics (Phase 15)',
38
+ verification_failed: 'sense of time (Phase 11)',
39
+ novel_dialog: 'reflexes (Phase 12)',
40
+ no_plan: 'exploration (Phase 14)',
41
+ };
42
+
43
+ /** Faculties that exist. Empty until Phase 11 lands the first one. */
44
+ export const BUILT_FACULTIES = new Set();
45
+
46
+ function metricPaths(udid) {
47
+ const dir = store.deviceDir(udid);
48
+ return {
49
+ dir,
50
+ escalations: path.join(dir, 'escalations.jsonl'),
51
+ flows: path.join(dir, 'flows.jsonl'),
52
+ baselines: path.join(dir, 'baselines'),
53
+ };
54
+ }
55
+
56
+ export { metricPaths as paths };
57
+
58
+ /**
59
+ * Append one record. Bookkeeping must never be able to fail a flow, so this
60
+ * swallows — but it swallows loudly enough to be found: the reason lands in
61
+ * `lastWriteError`, which `simframe escalations` prints.
62
+ */
63
+ let lastWriteError = null;
64
+ export const writeError = () => lastWriteError;
65
+
66
+ export function appendJsonl(file, record) {
67
+ try {
68
+ fs.mkdirSync(path.dirname(file), { recursive: true });
69
+ fs.appendFileSync(file, `${JSON.stringify(record)}\n`);
70
+ return true;
71
+ } catch (err) {
72
+ lastWriteError = `${file}: ${err.message}`;
73
+ return false;
74
+ }
75
+ }
76
+
77
+ /**
78
+ * Read a JSONL log, tolerating a torn last line.
79
+ *
80
+ * Two processes appending is not forbidden here, and a half-written line is a
81
+ * thing to skip rather than a reason to report no history at all.
82
+ */
83
+ export function readJsonl(file, { limit } = {}) {
84
+ let raw;
85
+ try {
86
+ raw = fs.readFileSync(file, 'utf8');
87
+ } catch {
88
+ return [];
89
+ }
90
+ const out = [];
91
+ for (const line of raw.split('\n')) {
92
+ if (!line.trim()) continue;
93
+ try {
94
+ out.push(JSON.parse(line));
95
+ } catch {
96
+ /* a torn append, or something else's file. Skip the line, keep the log. */
97
+ }
98
+ }
99
+ return limit ? out.slice(-limit) : out;
100
+ }
101
+
102
+ export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
103
+ export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
104
+
105
+ /**
106
+ * Mark an error as an escalation with a reason, at the site that knows why.
107
+ *
108
+ * Only ever adds a property: the message and the error class stay exactly what
109
+ * they were, so nothing above can behave differently for having been told.
110
+ * That is the whole reason classification happens by tagging rather than by
111
+ * matching error strings at the boundary — a regexed message is a reason that
112
+ * silently becomes "unknown" the day somebody rewords it.
113
+ */
114
+ export function tag(err, reason, { candidates = [], tried = [] } = {}) {
115
+ if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
116
+ err.escalation = { reason, candidates, tried };
117
+ return err;
118
+ }
119
+
120
+ /** The tag an error carries, or null. */
121
+ export function escalationOf(err) {
122
+ const e = err?.escalation;
123
+ if (!e || !REASONS.includes(e.reason)) return null;
124
+ return e;
125
+ }
126
+
127
+ /**
128
+ * Steps whose failure is a verification failure rather than a perception one.
129
+ *
130
+ * An assert that did not hold, or a wait that timed out, is simframe saying
131
+ * "I could not confirm this" — which is `verification_failed`, and is a
132
+ * different question from not knowing what is on screen.
133
+ */
134
+ const VERIFYING_STEPS = new Set([
135
+ 'assert', 'assertText', 'assertGone', 'waitText', 'waitFor', 'settle',
136
+ ]);
137
+
138
+ /**
139
+ * The reason a thrown step is an escalation.
140
+ *
141
+ * A tagged error wins, because the site that threw it knew more than this
142
+ * does. Everything else maps by step, and the fallback is
143
+ * `verification_failed` — a step that threw is a step whose effect could not
144
+ * be confirmed. There is no "unknown" branch on purpose.
145
+ */
146
+ export function reasonForStepError(step, err) {
147
+ const tagged = escalationOf(err);
148
+ if (tagged) return tagged;
149
+ const action = step?.action;
150
+ if (action === 'confirm' || action === 'chooseAny') return { reason: 'novel_dialog', candidates: [], tried: [] };
151
+ if (VERIFYING_STEPS.has(action)) return { reason: 'verification_failed', candidates: [], tried: [] };
152
+ return { reason: 'verification_failed', candidates: [], tried: [] };
153
+ }
154
+
155
+ /**
156
+ * Which verdicts hand a decision back.
157
+ *
158
+ * `unverified` does not: it means "this action has not been seen here before",
159
+ * which is a fact about the graph and not a question for anybody. The other
160
+ * two are — an unexpected screen halts the flow, and a step that moved nothing
161
+ * leaves the agent to judge whether the flow really did what it says.
162
+ */
163
+ export const ESCALATING_VERDICTS = new Set(['unexpected-screen', 'no-visible-change']);
164
+
165
+ /** How `goto`/`flow run` refusals map. They refuse rather than guess, and the refusal is the hand-back. */
166
+ export const PLAN_REASONS = {
167
+ 'unknown-screen': 'unknown_screen',
168
+ 'no-identity': 'unknown_screen',
169
+ ambiguous: 'ambiguous_intent',
170
+ 'no-route': 'no_plan',
171
+ 'unreplayable-edge': 'no_plan',
172
+ 'unknown-flow': 'no_plan',
173
+ };
174
+
175
+ /** A short, bounded description of a candidate element, for the log. */
176
+ export function candidateOf(t) {
177
+ if (!t) return null;
178
+ if (typeof t === 'string') return { label: t.slice(0, 40) };
179
+ return {
180
+ label: typeof t.label === 'string' ? t.label.slice(0, 40) : null,
181
+ x: t.x ?? null,
182
+ y: t.y ?? null,
183
+ region: t.region ?? null,
184
+ score: t.score ?? null,
185
+ };
186
+ }
187
+
188
+ /**
189
+ * What screen this is, for free.
190
+ *
191
+ * Read off what is already on disk — the newest frame's layout hash, and the
192
+ * structural fingerprint screen memory filed under it. A perception pass here
193
+ * would make logging cost something, and an instrumentation phase that slows
194
+ * the thing it measures has broken its own numbers.
195
+ */
196
+ export function fingerprintNow(udid, screenmap) {
197
+ try {
198
+ const state = store.readJson(store.paths(udid).state);
199
+ if (!state?.layoutHash) return null;
200
+ const near = screenmap?.recallNearest?.(udid, state.layoutHash);
201
+ return near?.entry?.structuralHash ?? null;
202
+ } catch {
203
+ return null;
204
+ }
205
+ }
206
+
207
+ /**
208
+ * Write one escalation.
209
+ *
210
+ * `tokens_spent` is null and stays null: the schema asks for it, and simframe
211
+ * is on the other side of the model from the thing that counts tokens. A
212
+ * plausible number derived from output length would be a guess wearing a
213
+ * measurement's clothes, and every number in this project is supposed to say
214
+ * where it came from.
215
+ */
216
+ export function recordEscalation(udid, {
217
+ flowId = null,
218
+ stepIndex = null,
219
+ fingerprint = null,
220
+ reason,
221
+ candidates = [],
222
+ tried = [],
223
+ outcome = 'escalated_to_model',
224
+ modelTurns = 1,
225
+ wallMs = null,
226
+ detail = null,
227
+ } = {}) {
228
+ if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
229
+ if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
230
+ const record = {
231
+ timestamp: new Date().toISOString(),
232
+ flow_id: flowId,
233
+ step_index: stepIndex,
234
+ screen_fingerprint: fingerprint,
235
+ reason,
236
+ candidate_elements: candidates.map(candidateOf).filter(Boolean).slice(0, 8),
237
+ reflex_or_exploration_tried: tried,
238
+ outcome,
239
+ model_turns_spent: modelTurns,
240
+ tokens_spent: null,
241
+ wall_time_ms: wallMs,
242
+ detail: detail ? String(detail).slice(0, 200) : null,
243
+ };
244
+ appendJsonl(metricPaths(udid).escalations, record);
245
+ return record;
246
+ }
247
+
248
+ /** Write one flow record. Shape is docs/research/03-human-parity.md §1. */
249
+ export function recordFlow(udid, record) {
250
+ appendJsonl(metricPaths(udid).flows, record);
251
+ return record;
252
+ }
253
+
254
+ /**
255
+ * Assemble a flow record from what a run already knows.
256
+ *
257
+ * `model_turns` counts the call itself as one, then adds one per escalation
258
+ * the agent has to answer. That is the number the whole series is trying to
259
+ * drive down, so it is defined here rather than left to whoever reads the log.
260
+ */
261
+ export function flowRecordFrom({
262
+ flowId,
263
+ flowName = null,
264
+ udid,
265
+ startedAt,
266
+ wallMs,
267
+ stepsTaken,
268
+ totalSteps,
269
+ minSteps = null,
270
+ imagesSent = 0,
271
+ escalations = [],
272
+ verdicts = [],
273
+ completed,
274
+ }) {
275
+ const histogram = {};
276
+ for (const v of verdicts) if (v) histogram[v] = (histogram[v] ?? 0) + 1;
277
+ const misTaps = verdicts.filter((v) => ESCALATING_VERDICTS.has(v)).length;
278
+ return {
279
+ flow_id: flowId,
280
+ flow_name: flowName,
281
+ device: udid,
282
+ started_at: new Date(startedAt).toISOString(),
283
+ wall_time_ms: wallMs,
284
+ steps_taken: stepsTaken,
285
+ total_steps: totalSteps,
286
+ min_steps: minSteps,
287
+ step_ratio: minSteps ? Number((stepsTaken / minSteps).toFixed(3)) : null,
288
+ model_turns: 1 + escalations.filter((e) => e.outcome !== 'resolved_locally').length,
289
+ images_sent: imagesSent,
290
+ input_tokens: null,
291
+ output_tokens: null,
292
+ escalations: escalations.map((e) => ({ reason: e.reason, step_index: e.step_index, outcome: e.outcome })),
293
+ escalation_count: escalations.length,
294
+ mis_taps: misTaps,
295
+ verdict_histogram: histogram,
296
+ reflex_firings: [],
297
+ exploration_events: [],
298
+ completed: Boolean(completed),
299
+ wrong_action_taken: verdicts.includes('unexpected-screen'),
300
+ };
301
+ }
302
+
303
+ // ---------------------------------------------------------------- statistics
304
+
305
+ export function median(xs) {
306
+ const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
307
+ if (!s.length) return null;
308
+ const mid = s.length >> 1;
309
+ return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2;
310
+ }
311
+
312
+ /**
313
+ * Quartiles by the "exclusive median" convention: each half excludes the
314
+ * median of an odd-length sample. Named because IQR is meaningless without it.
315
+ */
316
+ export function quartiles(xs) {
317
+ const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
318
+ if (!s.length) return null;
319
+ const mid = s.length >> 1;
320
+ const lower = s.slice(0, mid);
321
+ const upper = s.length % 2 ? s.slice(mid + 1) : s.slice(mid);
322
+ const p25 = median(lower) ?? s[0];
323
+ const p75 = median(upper) ?? s[s.length - 1];
324
+ return { min: s[0], p25, p50: median(s), p75, max: s[s.length - 1], iqr: p75 - p25, n: s.length };
325
+ }
326
+
327
+ /**
328
+ * The p-th percentile, by nearest-rank on the sorted sample.
329
+ *
330
+ * Nearest-rank rather than interpolation: with the 5-50 samples a graph edge
331
+ * carries, an interpolated p95 invents a value between two observations, and
332
+ * every number here is supposed to be one that actually happened.
333
+ */
334
+ export function percentile(xs, p) {
335
+ const s = xs.filter((x) => Number.isFinite(x)).sort((a, b) => a - b);
336
+ if (!s.length) return null;
337
+ const rank = Math.ceil((p / 100) * s.length);
338
+ return s[Math.min(s.length - 1, Math.max(0, rank - 1))];
339
+ }
340
+
341
+ /** The mean that punishes one slow flow, which is why §1 asks for it. */
342
+ export function harmonicMean(xs) {
343
+ const s = xs.filter((x) => Number.isFinite(x) && x > 0);
344
+ if (!s.length) return null;
345
+ return s.length / s.reduce((acc, x) => acc + 1 / x, 0);
346
+ }
347
+
348
+ /**
349
+ * HPI, exactly as §1 defines it.
350
+ *
351
+ * `flows` are agent runs (flows.jsonl records); `baselines` maps flow name to
352
+ * a human baseline. A flow with no human baseline gets no HPI_time — reported
353
+ * as null, never as 1.0, because a missing denominator is not parity.
354
+ */
355
+ export function hpi({ flows, baselines = {} }) {
356
+ const byName = new Map();
357
+ for (const f of flows) {
358
+ if (!f?.flow_name) continue;
359
+ if (!byName.has(f.flow_name)) byName.set(f.flow_name, []);
360
+ byName.get(f.flow_name).push(f);
361
+ }
362
+
363
+ const perFlow = [...byName.entries()].map(([name, runs]) => {
364
+ const agent = quartiles(runs.map((r) => r.wall_time_ms));
365
+ const human = baselines[name]?.wall_time_ms ?? null;
366
+ const humanMedian = human?.p50 ?? null;
367
+ const stepRatios = runs.map((r) => r.step_ratio).filter((x) => Number.isFinite(x));
368
+ return {
369
+ flow: name,
370
+ runs: runs.length,
371
+ agent_ms: agent,
372
+ human_median_ms: humanMedian,
373
+ hpi_time: humanMedian && agent?.p50 ? Number((humanMedian / agent.p50).toFixed(3)) : null,
374
+ step_ratio: median(stepRatios),
375
+ completed: runs.filter((r) => r.completed).length,
376
+ wrong_action: runs.filter((r) => r.wrong_action_taken).length,
377
+ escalations: runs.reduce((acc, r) => acc + (r.escalation_count ?? 0), 0),
378
+ model_turns: median(runs.map((r) => r.model_turns)),
379
+ };
380
+ }).sort((a, b) => a.flow.localeCompare(b.flow));
381
+
382
+ const total = flows.length;
383
+ const clean = flows.filter((f) => f.completed && !f.wrong_action_taken).length;
384
+ const accuracy = total ? Number((clean / total).toFixed(3)) : null;
385
+ const times = perFlow.map((f) => f.hpi_time).filter((x) => Number.isFinite(x));
386
+ const hpiTime = harmonicMean(times);
387
+ return {
388
+ flows: perFlow,
389
+ overall: {
390
+ runs: total,
391
+ flows_measured: perFlow.length,
392
+ flows_with_human_baseline: times.length,
393
+ hpi_accuracy: accuracy,
394
+ hpi_time: hpiTime == null ? null : Number(hpiTime.toFixed(3)),
395
+ hpi: hpiTime == null || accuracy == null ? null : Number((accuracy * hpiTime).toFixed(3)),
396
+ step_ratio: median(perFlow.map((f) => f.step_ratio)),
397
+ model_turns_median: median(perFlow.map((f) => f.model_turns)),
398
+ },
399
+ };
400
+ }
401
+
402
+ /**
403
+ * The escalation dashboard.
404
+ *
405
+ * `avoidable` is §8's definition — an escalation whose reason maps to a
406
+ * faculty that is not built yet, or is built and still let it through. One
407
+ * consequence is worth stating rather than hiding: with no faculty built, the
408
+ * rate is 1.0 by construction and says nothing. The per-reason breakdown is
409
+ * the part that decides the next phase, and it is informative today.
410
+ */
411
+ export function breakdown(records) {
412
+ const byReason = {};
413
+ for (const r of REASONS) byReason[r] = 0;
414
+ const byScreen = new Map();
415
+ const byOutcome = {};
416
+ let avoidable = 0;
417
+ for (const r of records) {
418
+ if (!REASONS.includes(r?.reason)) continue;
419
+ byReason[r.reason] += 1;
420
+ byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
421
+ // Already avoided locally, so not avoidable by anything unbuilt.
422
+ if (r.outcome !== 'resolved_locally') avoidable += 1;
423
+ const key = r.screen_fingerprint ?? '(no fingerprint)';
424
+ byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
425
+ }
426
+ const total = records.filter((r) => REASONS.includes(r?.reason)).length;
427
+ return {
428
+ total,
429
+ by_reason: byReason,
430
+ by_outcome: byOutcome,
431
+ faculty: Object.fromEntries(
432
+ REASONS.filter((r) => byReason[r]).map((r) => [r, `${FACULTY[r]}${BUILT_FACULTIES.has(FACULTY[r]) ? ' [built]' : ''}`]),
433
+ ),
434
+ avoidable,
435
+ avoidable_escalation_rate: total ? Number((avoidable / total).toFixed(3)) : null,
436
+ top_screens: [...byScreen.entries()]
437
+ .sort((a, b) => b[1] - a[1])
438
+ .slice(0, 10)
439
+ .map(([fingerprint, count]) => ({ fingerprint, count })),
440
+ model_turns_spent: records.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
441
+ };
442
+ }
443
+
444
+ /**
445
+ * How much slower than the committed baseline fails the build.
446
+ *
447
+ * 25%, not the 10% research §1 proposed, and the number is traceable to a
448
+ * measurement rather than to taste. Three HPI measurements of *identical code*
449
+ * on the same device the same afternoon gave HPI_time 0.475, 0.413 and 0.371,
450
+ * and per flow the medians moved up to 18% between runs — a 10% gate would
451
+ * have failed on noise roughly half the time, and a gate that cries wolf gets
452
+ * ignored, which costs more than no gate. Two things narrow the band instead
453
+ * of loosening it further: the gate reads the median of three passes rather
454
+ * than one, and accuracy stays strict at any drop at all. Numbers in
455
+ * docs/BENCHMARKS.md under "What the gate is set to, and why".
456
+ */
457
+ export const TIME_REGRESSION = 0.25;
458
+
459
+ /**
460
+ * The HPI_time a gate should compare: the median across passes when a report
461
+ * has them, and the single measurement when it does not — so a baseline
462
+ * committed before passes existed still gates.
463
+ */
464
+ export const gateTime = (o) => o?.hpi_time_median_of_passes ?? o?.hpi_time ?? null;
465
+
466
+ /**
467
+ * Compare a measurement against the committed baseline.
468
+ *
469
+ * A pure function rather than a few lines inside the CI script, because an
470
+ * untested gate is this project's recurring failure: the packaging check and a
471
+ * fingerprint eval both shipped unable to fail, and both looked exactly like
472
+ * this. Returns the reasons it should fail — empty means pass.
473
+ */
474
+ export function gateAgainst(baseline, measured, { timeRegression = TIME_REGRESSION } = {}) {
475
+ const base = baseline?.overall ?? {};
476
+ const now = measured?.overall ?? measured ?? {};
477
+ const baseTime = gateTime(base);
478
+ const nowTime = gateTime(now);
479
+ const failures = [];
480
+ if (base.hpi_accuracy != null && now.hpi_accuracy != null && now.hpi_accuracy < base.hpi_accuracy) {
481
+ failures.push(`HPI_accuracy dropped: ${now.hpi_accuracy} < ${base.hpi_accuracy} (any drop fails)`);
482
+ }
483
+ if (baseTime != null && nowTime != null && nowTime < baseTime * (1 - timeRegression)) {
484
+ failures.push(
485
+ `HPI_time regressed >${timeRegression * 100}%: ${nowTime} < ${(baseTime * (1 - timeRegression)).toFixed(3)}`,
486
+ );
487
+ }
488
+ // A checkout missing the human baseline the committed number was computed
489
+ // against would otherwise pass by having nothing to compare.
490
+ if (baseTime != null && nowTime == null) {
491
+ failures.push('HPI_time is null but the baseline has one — the human baseline it needs is missing from this checkout');
492
+ }
493
+ return failures;
494
+ }
495
+
496
+ /** A flow id that sorts by time and is short enough to read in a log. */
497
+ export function newFlowId() {
498
+ return `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
499
+ }
package/src/navigate.js CHANGED
@@ -6,6 +6,8 @@
6
6
  // somebody already walked, saved so it can be walked again.
7
7
  import { runScript } from './actions.js';
8
8
  import * as graph from './graph.js';
9
+ import * as metrics from './metrics.js';
10
+ import * as screenmap from './screenmap.js';
9
11
  import * as api from './index.js';
10
12
  import * as store from './store.js';
11
13
  import fs from 'node:fs';
@@ -31,6 +33,31 @@ export function stepFor(edge) {
31
33
  return null;
32
34
  }
33
35
 
36
+ /**
37
+ * A refusal to act, written down.
38
+ *
39
+ * `goto` and `flow run` refuse rather than guess, and a refusal is exactly
40
+ * "the tool handed the decision back" — the thing the escalation log exists to
41
+ * count. The reason mapping lives in metrics.PLAN_REASONS so the five reasons
42
+ * have one owner.
43
+ */
44
+ function refuse(udid, result, { detail = null } = {}) {
45
+ try {
46
+ const reason = metrics.PLAN_REASONS[result.reason];
47
+ if (reason) {
48
+ metrics.recordEscalation(udid, {
49
+ reason,
50
+ fingerprint: metrics.fingerprintNow(udid, screenmap),
51
+ outcome: 'escalated_to_model',
52
+ detail: detail ?? result.reason,
53
+ });
54
+ }
55
+ } catch {
56
+ /* a log that cannot be written must not change what is returned */
57
+ }
58
+ return result;
59
+ }
60
+
34
61
  /**
35
62
  * Walk to a known screen.
36
63
  *
@@ -43,8 +70,8 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
43
70
  const udid = device.udid;
44
71
 
45
72
  const found = graph.findScreen(udid, target);
46
- if (!found) return { ok: false, reason: 'unknown-screen', known: knownScreens(udid) };
47
- if (found.ambiguous) return { ok: false, reason: 'ambiguous', candidates: found.ambiguous };
73
+ if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"` });
74
+ if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens` });
48
75
 
49
76
  const here = await api.screenIdentity(udid, {});
50
77
  if (here.hash === found.node.hash) {
@@ -57,12 +84,12 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
57
84
  // `Cannot read properties of null (reading 'slice')` instead of answering.
58
85
  // Not hypothetical on Android, where README's own table puts the launcher at
59
86
  // one token.
60
- if (!here.hash) return { ok: false, reason: 'no-identity', to: found.name };
87
+ if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name });
61
88
  const path_ = graph.route(udid, { hash: here.hash, tokens: here.tokens }, found.node.hash);
62
- if (!path_) return { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name };
89
+ if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name });
63
90
 
64
91
  const steps = path_.map(stepFor);
65
- if (steps.some((s) => !s)) return { ok: false, reason: 'unreplayable-edge', to: found.name };
92
+ if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name });
66
93
 
67
94
  const result = await runScript(udid, { steps, stopOnUnexpected: true, ...runOptions });
68
95
  const arrived = await api.screenIdentity(udid, {});
@@ -121,7 +148,17 @@ export function listFlows(udid) {
121
148
  export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}) {
122
149
  const { device } = await api.ensureDaemon(deviceQuery, options);
123
150
  const flow = loadFlow(device.udid, name);
124
- if (!flow) return { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) };
125
- const result = await runScript(device.udid, { steps: flow.steps, stopOnUnexpected: true, ...runOptions });
151
+ if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"` });
152
+ // A replayed flow knows its own name, so its record can be compared against
153
+ // a human doing the same thing. `minSteps` comes from the flow definition or
154
+ // stays null — the step count of a recorded route is not a claim about the
155
+ // shortest one.
156
+ const result = await runScript(device.udid, {
157
+ steps: flow.steps,
158
+ stopOnUnexpected: true,
159
+ flowName: name,
160
+ minSteps: flow.minSteps ?? null,
161
+ ...runOptions,
162
+ });
126
163
  return { ok: result.ranSteps === flow.steps.length, name, ...result };
127
164
  }
@@ -176,6 +176,18 @@ async function resolveDevice(query, opts) {
176
176
  : 'no booted emulator (start one with `emulator -avd <name>`)',
177
177
  );
178
178
  }
179
+ // See the same guard in ios.js: an arbitrary pick is a tap on the wrong
180
+ // device, and this backend can act too.
181
+ if (booted.length > 1) {
182
+ throw Object.assign(
183
+ new Error(
184
+ `${booted.length} emulators are running and none was named: ` +
185
+ `${booted.map((d) => `${d.name} (${d.udid})`).join(', ')} — name one with --device, ` +
186
+ 'or set SIMFRAME_DEVICE to pick a default for this shell',
187
+ ),
188
+ { ambiguous: true },
189
+ );
190
+ }
179
191
  return booted[0];
180
192
  }
181
193
  const q = loose(query);
@@ -962,6 +974,16 @@ function toolchain() {
962
974
  * driver's name. `uiautomator dump` costs 2,012 ms a read, which is why it is
963
975
  * not the answer; see docs/DEFERRED.md for the shape of the one that would be.
964
976
  */
977
+ async function bootedAt(serial) {
978
+ try {
979
+ const out = await adb(serial, ['shell', 'cat', '/proc/uptime']);
980
+ const seconds = Number(String(out.stdout ?? out).trim().split(/\s+/)[0]);
981
+ return Number.isFinite(seconds) ? Date.now() - seconds * 1000 : null;
982
+ } catch {
983
+ return null;
984
+ }
985
+ }
986
+
965
987
  function capabilities() {
966
988
  return {
967
989
  captureEngines: ['screenshot'],
@@ -992,6 +1014,11 @@ export const platform = {
992
1014
  setPasteboard,
993
1015
  getPasteboard,
994
1016
  permissionServices: () => PERMISSION_SERVICES,
1017
+ // Uptime, because there is no CoreSimulator directory to stat. `/proc/uptime`
1018
+ // is seconds since boot, so boot is now minus that — and it is a real
1019
+ // answer rather than the other platform's vocabulary, which is the rule a
1020
+ // backend that cannot answer has to follow.
1021
+ bootedAt,
995
1022
  capabilities,
996
1023
  toolchain,
997
1024
  };
@@ -63,6 +63,7 @@ export const PLATFORM_SURFACE = Object.freeze([
63
63
  'geometry', 'inputDriver',
64
64
  'screenshot', 'launchApp', 'terminateApp', 'openUrl',
65
65
  'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
66
+ 'bootedAt',
66
67
  ]);
67
68
 
68
69
  /** @type {Record<string, Platform>} */
@@ -237,6 +238,8 @@ export function permissionServices(udid) {
237
238
  * knows. Tap points computed from that are wrong, and nothing says so.
238
239
  */
239
240
  export const geometryFor = (udid) => platformFor(udid).geometry(udid);
241
+ /** When the device last booted, epoch ms, or null. Both backends answer; neither guesses. */
242
+ export const bootedAtFor = (udid) => platformFor(udid).bootedAt(udid);
240
243
 
241
244
  /**
242
245
  * The backend's own input path, or null when input comes from above the