simframe 0.11.0 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,107 @@
1
+ #!/usr/bin/env node
2
+ // Score a ruling population against the baselines that could embarrass it.
3
+ //
4
+ // Written **before** the population it first scored was finished, deliberately.
5
+ // The previous attempt computed a `stillMs` threshold after seeing the answers,
6
+ // on 14 samples of which 12 shared one label, and produced "100%" — a number
7
+ // that was fitted, not measured. Fixing that afterwards is not possible: you
8
+ // cannot un-see the data. So the split rule, the candidate thresholds and the
9
+ // baselines are all fixed here in advance.
10
+ //
11
+ // node scripts/score-rulings.mjs --device=<udid>
12
+ //
13
+ // What it will not do: pick the best threshold and report its score. It fits on
14
+ // the first half and scores on the second, and prints both numbers so a gap
15
+ // between them is visible as overfitting rather than hidden as success.
16
+ import * as metrics from '../src/metrics.js';
17
+ import { resolveDevice } from '../src/platform/index.js';
18
+
19
+ const arg = (n, d) => {
20
+ const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
21
+ return hit ? hit.slice(n.length + 3) : d;
22
+ };
23
+
24
+ /** The fixtures, and the word each one's situation actually calls for. */
25
+ const CORRECT = {
26
+ 'the list is still loading; its rows arrive shortly after launch': 'wait',
27
+ 'the detail screen is still fetching; its text arrives shortly': 'wait',
28
+ 'the list arrives in waves and this row is in the last one': 'wait',
29
+ 'Review is blocked until Species is filled in, and it is empty': 'stop',
30
+ 'the first submit always fails and the second works, so waiting cannot help': 'stop',
31
+ };
32
+
33
+ /** `retry` and `wait` differ only in how long they settle, so both satisfy a wait. */
34
+ const satisfies = (decision, want) =>
35
+ (want === 'wait' ? decision === 'wait' || decision === 'retry' : decision === want);
36
+
37
+ const dev = await resolveDevice(arg('device'));
38
+ const all = metrics.readSupervisions(dev.udid).filter((r) => CORRECT[r.expect]);
39
+ // `--last=N` scores one batch rather than the whole log. Needed the first time
40
+ // this ran: the log still held rulings from a deliberately skewed population,
41
+ // so the balance read 19/35 and the baseline 65% when the batch actually under
42
+ // test was 16/16. Mixing a known-bad population into the denominator is the
43
+ // same error as before wearing a different hat.
44
+ const lastN = Number(arg('last', 0));
45
+ const rows = lastN > 0 ? all.slice(-lastN) : all;
46
+ if (rows.length < 8) {
47
+ console.error(`only ${rows.length} labelled ruling(s) — run scripts/collect-rulings.mjs first`);
48
+ process.exit(2);
49
+ }
50
+
51
+ const want = (r) => CORRECT[r.expect];
52
+ const counts = rows.reduce((a, r) => ({ ...a, [want(r)]: (a[want(r)] ?? 0) + 1 }), {});
53
+ const commonest = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
54
+ const baseline = commonest[1] / rows.length;
55
+
56
+ console.log(`${rows.length} labelled rulings on ${dev.name}`);
57
+ console.log(`balance: ${JSON.stringify(counts)}`);
58
+ console.log(`majority-class baseline: always "${commonest[0]}" scores ${(100 * baseline).toFixed(0)}%`);
59
+ if (baseline > 0.65) {
60
+ console.log('\nSKEWED — any accuracy below is mostly a fact about the fixture set.');
61
+ }
62
+
63
+ const acc = (rs, predict) => rs.filter((r) => satisfies(predict(r), want(r))).length / (rs.length || 1);
64
+ const pct = (x) => `${(100 * x).toFixed(0)}%`;
65
+
66
+ console.log(`\n${'the model'.padEnd(30)} ${pct(acc(rows, (r) => r.decision))}`);
67
+ console.log(`${`always "${commonest[0]}"`.padEnd(30)} ${pct(baseline)}`);
68
+
69
+ // --- the stillMs rule, fitted on one half and scored on the other.
70
+ //
71
+ // Split by *recording order* rather than at random or by fixture, because the
72
+ // alternative is choosing a split, and choosing is the thing that went wrong.
73
+ const half = Math.floor(rows.length / 2);
74
+ const fit = rows.slice(0, half);
75
+ const held = rows.slice(half);
76
+ const CANDIDATES = [1000, 1500, 2000, 2500, 3000, 4000, 5000, 6000, 7000];
77
+ const rule = (t) => (r) => ((r.still_ms ?? 0) > t ? 'stop' : 'wait');
78
+
79
+ let best = null;
80
+ for (const t of CANDIDATES) {
81
+ const a = acc(fit, rule(t));
82
+ if (!best || a > best.a) best = { t, a };
83
+ }
84
+ console.log(`\nstillMs threshold, fitted on the first ${fit.length} and scored on the last ${held.length}:`);
85
+ console.log(` chosen on the fit half: > ${best.t}ms (${pct(best.a)} there)`);
86
+ console.log(` ${'on the held-out half:'.padEnd(24)} ${pct(acc(held, rule(best.t)))}`);
87
+ console.log(` ${'the model, same half:'.padEnd(24)} ${pct(acc(held, (r) => r.decision))}`);
88
+ console.log('\nA rule that scores far better on the fit half than the held-out half was');
89
+ console.log('fitted to noise. That gap is the number this script exists to print.');
90
+
91
+ // --- the direction of the errors, which does not depend on the balance at all.
92
+ const wrong = rows.filter((r) => !satisfies(r.decision, want(r)));
93
+ const dirs = wrong.reduce((a, r) => {
94
+ const k = `said "${r.decision}" where "${want(r)}" was right`;
95
+ return { ...a, [k]: (a[k] ?? 0) + 1 };
96
+ }, {});
97
+ console.log(`\n${wrong.length} error(s) of ${rows.length}:`);
98
+ for (const [k, n] of Object.entries(dirs).sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${k}`);
99
+ if (Object.keys(dirs).length === 1 && wrong.length > 2) {
100
+ console.log('\nAll errors in one direction. A one-directional bias is what an abstain');
101
+ console.log('token addresses (DEFERRED 100), whatever the headline accuracy says.');
102
+ }
103
+
104
+ const lat = rows.map((r) => r.latency_ms).filter(Number.isFinite).sort((a, b) => a - b);
105
+ if (lat.length) console.log(`\nmedian judgement latency: ${lat[lat.length >> 1]}ms`);
106
+ const timed = rows.filter((r) => Number.isFinite(r.edge_p95_ms)).length;
107
+ console.log(`edges the graph had timed: ${timed}/${rows.length}`);
@@ -0,0 +1,72 @@
1
+ #!/usr/bin/env node
2
+ // How long can this device be driven before capture wedges?
3
+ //
4
+ // The wedge is the thing standing between us and a ruling population: the
5
+ // display stops rendering after a few minutes of hard driving, `simctl
6
+ // screenshot` fails too, both of the daemon's recoveries fail, and only a
7
+ // device restart cures it. Three times in one afternoon.
8
+ //
9
+ // A leak was found in our own code on 2026-09-12 — damage-callback registration
10
+ // was not idempotent, and the recovery loop called it on every attempt, so one
11
+ // log's 670 port re-resolves meant up to 670 live callbacks on one port, each
12
+ // invoked per redraw. That is a real leak with a plausible path to saturating
13
+ // the display service, and it is **not proof** that it is the cause. This is
14
+ // how we find out: drive until it dies, and report how long that took.
15
+ //
16
+ // node scripts/soak-capture.mjs --device=<udid> --minutes=25
17
+ //
18
+ // A number to compare against, not a pass/fail. Before the fix, the device
19
+ // wedged roughly every 10-20 minutes of this kind of work.
20
+ import * as actions from '../src/actions.js';
21
+ import * as api from '../src/index.js';
22
+ import * as store from '../src/store.js';
23
+
24
+ const arg = (n, d) => {
25
+ const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
26
+ return hit ? hit.slice(n.length + 3) : d;
27
+ };
28
+ const device = arg('device');
29
+ const minutes = Number(arg('minutes', 20));
30
+ const BUNDLE = 'com.example.simframetestbed';
31
+
32
+ const { device: dev } = await api.ensureDaemon(device);
33
+ console.log(`device: ${dev.name} (${dev.runtime}) budget: ${minutes} minutes`);
34
+
35
+ const started = Date.now();
36
+ const deadline = started + minutes * 60_000;
37
+ let laps = 0;
38
+ let reads = 0;
39
+ const elapsed = () => ((Date.now() - started) / 60_000).toFixed(1);
40
+
41
+ // A lap is deliberately the kind of work that provokes it: launching, moving
42
+ // between screens, and a cold read on each — not idling with a poll.
43
+ const LAP = [
44
+ [{ launch: { value: BUNDLE, relaunch: true } }, { pause: 2500 }],
45
+ [{ tap: 'Forms, tab, 2 of 3' }, { pause: 800 }],
46
+ [{ tap: 'Long form' }, { pause: 900 }],
47
+ [{ scroll: 'down' }, { pause: 500 }],
48
+ [{ scroll: 'down' }, { pause: 500 }],
49
+ [{ tap: 'Plants, tab, 1 of 3' }, { pause: 900 }],
50
+ [{ tap: 'Diagnostics, tab, 3 of 3' }, { pause: 900 }],
51
+ ];
52
+
53
+ while (Date.now() < deadline) {
54
+ for (const steps of LAP) {
55
+ try {
56
+ await actions.runScript(device, { steps, verify: false, options: { supervisor: 'none' } });
57
+ } catch { /* a failed step is not the subject; a dead display is */ }
58
+ try {
59
+ await api.screenIdentity(device, { fresh: true, confirmNovel: false });
60
+ reads += 1;
61
+ } catch (err) {
62
+ console.log(`\nWEDGED after ${elapsed()} minutes, ${laps} laps, ${reads} cold reads`);
63
+ console.log(` ${String(err.message).split('\n')[0]}`);
64
+ const health = store.captureHealth(dev.udid);
65
+ if (health) console.log(` captureHealth: ${JSON.stringify(health)}`);
66
+ process.exit(1);
67
+ }
68
+ }
69
+ laps += 1;
70
+ if (laps % 5 === 0) process.stdout.write(` ${elapsed()}m ${laps} laps ${reads} reads still alive\n`);
71
+ }
72
+ console.log(`\nSURVIVED ${minutes} minutes: ${laps} laps, ${reads} cold reads, no wedge`);
@@ -230,7 +230,10 @@ Measured against the same screen as an image: **~460 tokens of text versus
230
230
  image path degrades to base64-as-text. The text also says what is *tappable*
231
231
  and where, which an image does not.
232
232
 
233
- The numbers are selectors. Whatever `ui` calls `#3`, you can tap as `#3`.
233
+ The numbers are selectors, and so are the labels beside them. Act by **name** —
234
+ whatever `ui` calls `Weekly digest`, you can tap as `"Weekly digest"`. A `#3`
235
+ is exact but only until the screen moves; see Selectors below for why that
236
+ order is a correction rather than a preference.
234
237
 
235
238
  **Reach for an image only when the text genuinely cannot answer the question:**
236
239
  visual layout, colour, spacing, an animation, or something neither the
@@ -246,7 +249,7 @@ a baseline captured *before* it, so steps cannot race the UI.
246
249
  cat > /tmp/flow.json <<'JSON'
247
250
  [{"tap": "Inbox tab"},
248
251
  {"assert": {"value": "Weekly digest", "is": "visible"}},
249
- {"tap": "#3"},
252
+ {"tap": "Weekly digest"},
250
253
  {"type": {"into": "Reply", "text": "on it"}},
251
254
  {"scrollTo": "Send"},
252
255
  {"tap": "Send"},
@@ -312,6 +315,9 @@ the next call should change.
312
315
  | `#N cannot be trusted here — … this is a different screen` | the screen really did change. Read it again |
313
316
  | `autoSettle was off, so the map below was read without waiting` | the map may describe the screen *before* the last action landed |
314
317
  | `"iPhone 17 Pro" names more than one booted device` | a name cannot identify which device answered. Pass `device` with a UDID |
318
+ | `frame 2.5s old` | the map was read from a frame that old. Usually fine after a deliberate pause; worth noticing if you expected to have just acted |
319
+ | `WARNING this frame is Ns old — … a description of the past` | the screen may have moved on entirely. **Pass `refresh`** before believing any element below it. Reported from the field against 0.11.0: a complete 20-element map of a screen the app was not on, with nothing saying so |
320
+ | `capture is wedged and both recoveries are spent` | the simulator has stopped rendering and neither of simframe's recoveries helped. Run `simframe revive --device=<udid>`; nothing else will work until you do |
315
321
 
316
322
  ## Every step is verified, and the verdict means something
317
323
 
@@ -388,8 +394,19 @@ as one call adds one exchange to the context instead of twelve.
388
394
  ```bash
389
395
  simframe doctor # capture engine, input driver, a11y, OCR — each honestly
390
396
  simframe doctor --strict # any degraded layer is a non-zero exit
397
+ simframe input reset # rebuild the HID session, without restarting anything
398
+ simframe revive # power-cycle a device that has stopped rendering
391
399
  ```
392
400
 
401
+ **If capture has wedged**, `doctor` says so and nothing below it will work. A
402
+ simulator driven hard for several minutes can stop rendering — `simctl
403
+ screenshot` fails too, so it is the device and not simframe's view of it — and
404
+ the daemon tries two recoveries, reports the state, and stops there. `simframe
405
+ revive` does the restart in the order that matters and confirms frames are
406
+ flowing again afterwards. It is a command rather than a behaviour on purpose:
407
+ a capture loop that rebooted the device it was watching would be a tool
408
+ reaching for the mains.
409
+
393
410
  simframe falls back when it must — the simctl capture loop instead of the
394
411
  daemon, idb instead of the in-process input and accessibility paths — but it
395
412
  never falls back quietly. If
@@ -405,6 +422,7 @@ simframe recall # what happened in the last ~60s, as text
405
422
  simframe strip # recent frames tiled into one image, for an animation
406
423
  simframe find "the save button" # resolve an intent without acting on it
407
424
  simframe wait --mode=settle # block until the screen stops reacting
425
+ simframe supervisions # what the local supervisor decided, and what came of it
408
426
  ```
409
427
 
410
428
  `recall` matters more than it looks: if you look up and the screen is already
package/src/actions.js CHANGED
@@ -209,6 +209,62 @@ export async function runScript(
209
209
  /* instrumentation must not be able to fail a flow it is only watching */
210
210
  }
211
211
  };
212
+ /**
213
+ * What the caller is shown for each outcome, keyed by what the log stores.
214
+ *
215
+ * Two vocabularies on purpose. The log wants tokens that will still parse in
216
+ * six months; the result and the CLI line want English, and a test already
217
+ * pins those words. Mapping them here is what stops the two drifting.
218
+ */
219
+ const SHOWN = {
220
+ recovered: 'recovered',
221
+ still_failed: 'still failed',
222
+ stopped: 'stopped the run',
223
+ no_ruling: 'no ruling',
224
+ };
225
+ /**
226
+ * Record one ruling, in memory for the caller and on disk for the analysis.
227
+ *
228
+ * One helper rather than four call sites because the two had already drifted
229
+ * apart once — `from` was on the `stop` push and on none of the others, so a
230
+ * rule-sourced wait was indistinguishable from a model-sourced one in the
231
+ * only place that reported them. A ruling that reaches the caller and not the
232
+ * log is the state item 101a exists to end: three rulings had ever existed
233
+ * anywhere, and 101, 106 and 96 are all waiting on a population.
234
+ */
235
+ const noteRuling = (index, ruling, outcome, { step, expect, failure, why } = {}) => {
236
+ const shown = SHOWN[outcome] ?? outcome;
237
+ const unanswered = why?.kind
238
+ ? `the supervisor did not answer: ${why.kind}`
239
+ : 'the supervisor did not answer';
240
+ supervisions.push({
241
+ index,
242
+ decision: ruling?.decision ?? 'unavailable',
243
+ reason: ruling?.reason ?? unanswered,
244
+ from: ruling?.from,
245
+ outcome: shown,
246
+ });
247
+ try {
248
+ metrics.recordSupervision(udid, {
249
+ index,
250
+ step: step ? graph.actionSignature(step) : ruling?.context?.action ?? null,
251
+ edge: ruling?.context?.edge ?? null,
252
+ screen: ruling?.context?.screen ?? null,
253
+ decision: ruling?.decision ?? 'unavailable',
254
+ from: ruling?.from ?? (ruling ? 'model' : 'none'),
255
+ reason: ruling?.reason ?? unanswered,
256
+ ms: ruling?.ms,
257
+ stillMs: ruling?.context?.stillMs,
258
+ p95: ruling?.context?.p95,
259
+ samples: ruling?.context?.samples,
260
+ expect,
261
+ failure,
262
+ outcome,
263
+ });
264
+ } catch {
265
+ /* instrumentation must not be able to fail a flow it is only watching */
266
+ }
267
+ };
212
268
 
213
269
  const needsInput = steps.some((s) => ACTION_STEPS.has(normalizeStep(s).action));
214
270
  if (needsInput) {
@@ -309,8 +365,15 @@ export async function runScript(
309
365
  // Ask the supervisor before anything is abandoned. It sits behind the
310
366
  // hands and in front of the reasoner: first responder, not
311
367
  // decision-maker, and its whole vocabulary is wait/retry/stop.
368
+ // Why a consultation produced nothing, for the log. Every failure used
369
+ // to arrive as "the supervisor did not answer" — a timeout, a guardrail
370
+ // refusal and a model that was never installed reading identically.
371
+ const why = {};
312
372
  const ruling = await superviseFailure(deviceQuery, {
313
- goal: supervise ?? flowName, step, expected: step.expect, err, options,
373
+ goal: supervise ?? flowName, step, expected: step.expect, err, options, udid, detail: why,
374
+ });
375
+ const ruled = (outcome) => noteRuling(i, ruling, outcome, {
376
+ step, expect: step.expect, failure: err.message, why,
314
377
  });
315
378
  if (ruling?.decision === 'wait' || ruling?.decision === 'retry') {
316
379
  // Both wait and retry settle first, differing only in how long.
@@ -332,10 +395,10 @@ export async function runScript(
332
395
  try {
333
396
  detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
334
397
  detail += ` [the local supervisor said ${ruling.decision}; it worked on the second attempt]`;
335
- supervisions.push({ index: i, decision: ruling.decision, reason: ruling.reason, outcome: 'recovered' });
398
+ ruled('recovered');
336
399
  continue;
337
400
  } catch (again) {
338
- supervisions.push({ index: i, decision: ruling.decision, reason: ruling.reason, outcome: 'still failed' });
401
+ ruled('still_failed');
339
402
  err = again;
340
403
  }
341
404
  } else if (!ruling && supervisor.requested(options)) {
@@ -345,10 +408,12 @@ export async function runScript(
345
408
  // model healthy. The reporter's own words — *"had I run flow 2 alone
346
409
  // I would have reported the supervisor makes no difference without
347
410
  // realising it had never run"*.
348
- supervisions.push({ index: i, decision: 'unavailable', reason: 'the supervisor did not answer', outcome: 'no ruling' });
349
- err.message += ' — the local supervisor was consulted and did not answer, so this failure was not judged.';
411
+ ruled('no_ruling');
412
+ err.message += ' — the local supervisor was consulted and did not answer'
413
+ + (why.kind ? ` (${why.kind})` : '')
414
+ + ', so this failure was not judged.';
350
415
  } else if (ruling?.decision === 'stop') {
351
- supervisions.push({ index: i, decision: 'stop', reason: ruling.reason, from: ruling.from, outcome: 'stopped the run' });
416
+ ruled('stopped');
352
417
  const remaining = steps.slice(i);
353
418
  err.message += ` — the local supervisor stopped the run here.`
354
419
  + ` ${remaining.length} step(s) were not attempted.`
@@ -1123,14 +1188,34 @@ const SUPERVISOR_RETRY_MS = 900;
1123
1188
  /** A scroll moves at once or not at all; it does not need a transition's budget. */
1124
1189
  const SCROLL_SETTLE_MS = 800;
1125
1190
 
1126
- async function superviseFailure(deviceQuery, { goal, step, expected, err, options }) {
1191
+ async function superviseFailure(deviceQuery, { goal, step, expected, err, options, udid, detail }) {
1127
1192
  if (!supervisor.requested(options)) return null;
1193
+ // The action half of the edge is free — the step is in hand — so a rule-sourced
1194
+ // ruling still records which action it was about even though it never reads
1195
+ // the screen. Deliberately no `screenMap` call on this path: the rule exists
1196
+ // to answer without one, and paying for a read to make the log prettier would
1197
+ // slow the fast path to improve the bookkeeping.
1198
+ const action = graph.actionSignature(step);
1128
1199
  const settled = deterministicRuling(err);
1129
- if (settled) return { decision: settled.decision, reason: settled.why, from: 'rule' };
1200
+ if (settled) {
1201
+ return {
1202
+ decision: settled.decision, reason: settled.why, from: 'rule',
1203
+ context: { action, edge: action, screen: null, stillMs: null, p95: null, samples: null },
1204
+ };
1205
+ }
1130
1206
  try {
1131
1207
  const map = await view.screenMap(deviceQuery, { options, refresh: false });
1132
1208
  const stillMs = map.identity?.state?.motion?.stillForMs;
1133
- return await supervisor.judge({
1209
+ const screen = map.identity?.hash ?? null;
1210
+ // What the graph knew about this edge at the moment of the ruling. Read
1211
+ // here and not at analysis time on purpose: the graph keeps learning, so a
1212
+ // p95 looked up next week is not the number this ruling was competing
1213
+ // with, and item 101 asks exactly "would a p95 lookup have got this right".
1214
+ let timing = null;
1215
+ try {
1216
+ timing = udid && map.identity ? graph.timingFor(udid, map.identity, step) : null;
1217
+ } catch { /* an unknown edge is a fact about the graph, not a failure here */ }
1218
+ const ruling = await supervisor.judge({
1134
1219
  goal,
1135
1220
  step: `${step.action} ${JSON.stringify(String(step.value ?? step.target ?? step.into ?? step.seek ?? '').slice(0, 60))}`,
1136
1221
  expected,
@@ -1139,7 +1224,21 @@ async function superviseFailure(deviceQuery, { goal, step, expected, err, option
1139
1224
  stillMs,
1140
1225
  note: stillFillingIn(map.identity?.entry),
1141
1226
  options,
1227
+ detail,
1142
1228
  });
1229
+ if (!ruling) return null;
1230
+ return {
1231
+ ...ruling,
1232
+ from: ruling.from ?? 'model',
1233
+ context: {
1234
+ action,
1235
+ edge: screen ? `${String(screen).slice(0, 12)}:${action}` : action,
1236
+ screen,
1237
+ stillMs,
1238
+ p95: timing?.p95 ?? null,
1239
+ samples: timing?.samples ?? null,
1240
+ },
1241
+ };
1143
1242
  } catch {
1144
1243
  return null;
1145
1244
  }
package/src/cli.js CHANGED
@@ -3,7 +3,7 @@ import fs from 'node:fs';
3
3
  import os from 'node:os';
4
4
  import path from 'node:path';
5
5
  import { runDaemon, DEFAULTS } from './daemon.js';
6
- import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
6
+ import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, restartDevice, screenshot, toolchainChecks } from './platform/index.js';
7
7
  import * as actions from './actions.js';
8
8
  import * as analyze from './analyze.js';
9
9
  import * as api from './index.js';
@@ -47,6 +47,8 @@ const USAGE = `simframe — always-warm iOS Simulator frames
47
47
  simframe baseline list recorded runs per flow, and what is committed
48
48
  simframe hpi [device] Human Parity Index, per flow and overall
49
49
  simframe escalations [device] why simframe handed decisions back, by reason
50
+ simframe supervisions [device] local supervisor rulings, and what came of each
51
+ simframe revive [device] power-cycle a wedged device: stop, shutdown, boot, start, reset input
50
52
  (--session=<id> narrows to one agent; the
51
53
  ids are listed in the output. SIMFRAME_SESSION
52
54
  names one, but only at process start — an
@@ -56,10 +58,12 @@ const USAGE = `simframe — always-warm iOS Simulator frames
56
58
  (--strict, or SIMFRAME_STRICT=1, makes any
57
59
  degraded layer a non-zero exit)
58
60
 
59
- Selectors — anywhere a control is named
60
- #3 the number \`simframe ui\` gave it. Cheapest, unambiguous.
61
- "Save" a label or a phrase, resolved by intent (verbs, typos, synonyms)
62
- @120,400 raw point coordinates
61
+ Selectors — anywhere a control is named, best first
62
+ "Save" a label or a phrase, resolved by intent (verbs, typos, synonyms,
63
+ icon-only controls by their common name). Start here.
64
+ #3 the number \`simframe ui\` gave it. Exact, but only inside the
65
+ round trip that numbered it — the screen moves and it does not.
66
+ @120,400 raw point coordinates. Last resort: it cannot tell you it missed.
63
67
 
64
68
  Measuring against a human — the Human Parity Index
65
69
 
@@ -378,6 +382,51 @@ async function main() {
378
382
  return;
379
383
  }
380
384
 
385
+ // The power cycle, when the narrow remedies are spent.
386
+ //
387
+ // Deliberately a command and not a behaviour. The capture loop tries two
388
+ // things — re-resolve the display port, then rebind the device — and then
389
+ // reports `stalled` and stops, because a capture loop that rebooted the
390
+ // device it was watching would be a tool reaching for the mains when a
391
+ // reading looks wrong. Restarting is the operator's call.
392
+ //
393
+ // But it was the operator's call *and* their four commands, remembered from
394
+ // a handoff note: stop the daemon, shut the device down, boot it and wait,
395
+ // start capture, rebuild the HID session. Done by hand three times in one
396
+ // afternoon, in that order, because any other order leaves a daemon holding
397
+ // a dead device. So the tool knows the order now; the decision is still
398
+ // yours.
399
+ case 'revive': {
400
+ const dev = await resolveDevice(device);
401
+ const say = (line) => { if (!flags.json) console.log(line); };
402
+ const steps = [];
403
+ const did = async (what, fn) => {
404
+ try { await fn(); steps.push({ step: what, ok: true }); say(` ok ${what}`); } catch (err) {
405
+ steps.push({ step: what, ok: false, error: err.message });
406
+ say(` .. ${what} — ${err.message.split('\n')[0]}`);
407
+ }
408
+ };
409
+ say(`reviving ${dev.name}`);
410
+ // Forced: the point of this command is that the device is wedged, so
411
+ // something is certainly still holding it.
412
+ await did('stopped the daemon', async () => { api.stopDaemon(dev.udid, { force: true }); });
413
+ // Through the boundary, which is the whole point of the boundary: the
414
+ // first version of this shelled out to `xcrun` from here and the test
415
+ // that forbids it failed immediately, correctly.
416
+ await did('restarted the device, and waited for the boot to finish',
417
+ () => restartDevice(dev.udid));
418
+ await did('started capture', () => api.ensureDaemon(dev.udid));
419
+ await did('rebuilt the HID session', () => input.resetSession(dev.udid));
420
+ const health = await api.getState(dev.udid).then((s) => s?.state ?? null).catch(() => null);
421
+ const alive = Boolean(health?.hash);
422
+ emit(flags, { ok: alive, device: dev.udid, steps }, alive
423
+ ? `\n${dev.name} is producing frames again`
424
+ : `\n${dev.name} is still not producing frames. This is past what simframe can do —`
425
+ + ' check Simulator.app is not showing an error, and see docs/DEFERRED.md item 95.');
426
+ if (!alive) process.exitCode = 1;
427
+ return;
428
+ }
429
+
381
430
  case 'status': {
382
431
  const udids = device
383
432
  ? [(await resolveDevice(device)).udid]
@@ -1027,6 +1076,37 @@ async function main() {
1027
1076
  return;
1028
1077
  }
1029
1078
 
1079
+ case 'supervisions': {
1080
+ const dev = await resolveDevice(flags.device);
1081
+ const records = metrics.readSupervisions(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
1082
+ const b = metrics.supervisionBreakdown(records);
1083
+ if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify({ ...b, records }, null, 2)}\n`);
1084
+ emit(flags, { ...b, records: flags.verbose ? records : undefined }, [
1085
+ `${b.total} supervisor ruling${b.total === 1 ? '' : 's'} on ${dev.name}`,
1086
+ b.total ? '' : 'Nothing has been judged on this device yet. The supervisor is off unless'
1087
+ + ' SIMFRAME_SUPERVISOR=apple, and a ruling is only recorded when a step actually fails.',
1088
+ ...Object.entries(b.decision_to_outcome)
1089
+ .sort((a, c) => c[1] - a[1])
1090
+ .map(([k, n]) => ` ${k.padEnd(28)} ${String(n).padStart(4)}`),
1091
+ b.total ? '' : null,
1092
+ b.total ? `sourced: ${Object.entries(b.by_from).map(([k, n]) => `${k} ${n}`).join(', ')}` : null,
1093
+ b.median_latency_ms != null ? `median latency: ${b.median_latency_ms}ms` : null,
1094
+ // Said out loud, because the first version of item 101 claimed its
1095
+ // measurement ran "on logs we already have" when nothing persisted a
1096
+ // ruling at all. This line is what stops that claim being made twice.
1097
+ b.total
1098
+ ? `edges the graph had timed: ${b.p95_known}/${b.total}`
1099
+ + (b.p95_unknown
1100
+ ? ` — ${b.p95_unknown} ruling(s) are on edges with no p95, so they cannot take part in 101's comparison`
1101
+ : '')
1102
+ : null,
1103
+ b.sessions.length > 1
1104
+ ? `WARNING ${b.sessions.length} sessions are pooled here; two agents on one device write one file`
1105
+ : null,
1106
+ ].filter((l) => l !== null).join('\n'));
1107
+ break;
1108
+ }
1109
+
1030
1110
  case 'escalations': {
1031
1111
  const dev = await resolveDevice(flags.device);
1032
1112
  const records = metrics.readEscalations(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
@@ -48,8 +48,32 @@ import * as regions from './regions.js';
48
48
  * extends upward while the rows above keep being key-shaped, which page
49
49
  * content is not. Any screen fingerprinted with a keyboard up hashes
50
50
  * differently, so the stored graph and maps must go again.
51
+ *
52
+ * 7 — a screen whose own name is an iOS **large title** had no name at all in
53
+ * its identity. Chrome labels are the only text these tokens keep, and a
54
+ * large title is drawn tight against the content it heads — 79 pt of inset
55
+ * above it and 5.3 pt below, against a boundary bar of 66.5 — so the top
56
+ * chrome detector, which looks for the gap *beneath* a bar, never found it
57
+ * and the title was discarded as content. Measured on the Settings root in
58
+ * both sensor modes: **0 named tokens**, and the same for Contacts and
59
+ * Reminders. On a hosted runner two sparse nameless readings then matched
60
+ * exactly and one hash stood for two different screens. `regions.bands`
61
+ * now finds a large title by the inset above it; every affected screen
62
+ * hashes differently, so stored graphs and maps go again.
63
+ *
64
+ * 8 — the same rule, keyed on the wrong thing. Its "is this a real inset"
65
+ * test compared the gap above the title against the screen's **median row
66
+ * gap**, so whether a system-drawn title counted as chrome depended on how
67
+ * many rows happened to sit below it. Caught on the React Native testbed's
68
+ * first day, with two screens of one app: a list of 24 rows has a median
69
+ * gap of 0 and the rule fired, a list of 4 above a tab bar has a median gap
70
+ * of **414** and it did not. Same title, same 62.9pt inset, opposite
71
+ * answers — so one screen carried a name and the other did not, and the
72
+ * graph merged them. Both bounds are absolute now. Screens that were
73
+ * missed at 7 hash differently at 8, and unlike a stale hash that matches
74
+ * nothing, these matched the *wrong* thing.
51
75
  */
52
- export const TOKEN_RULES_VERSION = 6;
76
+ export const TOKEN_RULES_VERSION = 8;
53
77
 
54
78
  /** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
55
79
  export const GRID = 24;
package/src/graph.js CHANGED
@@ -258,6 +258,28 @@ export function nearestScreen(udid, screen, { threshold = SIMILARITY_THRESHOLD }
258
258
  for (const node of nodes) {
259
259
  for (const f of fingerprintsOf(node)) {
260
260
  if (!f.tokens?.length) continue;
261
+ // A screen that says it is called something else is not this screen.
262
+ //
263
+ // Similarity weighs every token equally, and a chrome label is not an
264
+ // equal token — it is the only text in a fingerprint and the only thing
265
+ // that distinguishes two screens of the same shape. `fingerprint.js` has
266
+ // said so in a comment since it was written: "two list screens with
267
+ // identical structure differ by their title, and nothing else says so".
268
+ // Nothing enforced it.
269
+ //
270
+ // Measured on the RN testbed, which is what found this: a list of plants
271
+ // and a list of forms, each a large title over full-width rows above a
272
+ // tab bar, scored **0.50** against a 0.36 threshold and became one node.
273
+ // Both were correctly named by then; the name was simply outvoted, being
274
+ // one token of a union of six. The graph then offered one screen's
275
+ // controls on the other, which is the failure `verify` exists to stop.
276
+ //
277
+ // Both sides must actually carry names for this to apply. A reading whose
278
+ // tree did not answer has no names through no fault of the screen's, and
279
+ // the two sensors are already known to disagree about 0.33-0.47 of a
280
+ // token set — so "one has names, the other does not" is a fact about the
281
+ // sensors and must not be read as a fact about identity.
282
+ if (disagreeOnName(f.tokens, key.tokens)) continue;
261
283
  const s = fingerprint.similarity(f.tokens, key.tokens);
262
284
  if (s > bestSimilarity) {
263
285
  bestSimilarity = s;
@@ -268,6 +290,31 @@ export function nearestScreen(udid, screen, { threshold = SIMILARITY_THRESHOLD }
268
290
  return best && bestSimilarity >= threshold ? { node: best, similarity: bestSimilarity } : null;
269
291
  }
270
292
 
293
+ /** The chrome labels in a token set — the only text a fingerprint keeps. */
294
+ function namesIn(tokens) {
295
+ const out = new Set();
296
+ for (const t of tokens ?? []) {
297
+ const open = t.indexOf('"');
298
+ if (open < 0) continue;
299
+ const close = t.lastIndexOf('"');
300
+ if (close > open) out.add(t.slice(open + 1, close));
301
+ }
302
+ return out;
303
+ }
304
+
305
+ /**
306
+ * Do these two readings name themselves, and name themselves differently?
307
+ *
308
+ * Only a positive disagreement counts. Silence on either side is not evidence.
309
+ */
310
+ function disagreeOnName(a, b) {
311
+ const left = namesIn(a);
312
+ const right = namesIn(b);
313
+ if (!left.size || !right.size) return false;
314
+ for (const name of left) if (right.has(name)) return false;
315
+ return true;
316
+ }
317
+
271
318
  /**
272
319
  * Teach a node that it also looks like this.
273
320
  *