simframe 0.12.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -164,8 +164,21 @@ for (let round = 1; round <= rounds; round += 1) {
164
164
  // means the navigation did not happen.
165
165
  const s = fingerprint.similarity(previous.tokens, id.tokens ?? []);
166
166
  if (s >= ARRIVAL_SUSPICION) {
167
+ // What was actually on screen, not just that it was the wrong thing.
168
+ //
169
+ // This check has fired twice on CI and both times the report was a
170
+ // similarity score and two screen names, which is enough to know the
171
+ // run is void and not enough to know why. The tour asserts arrival with
172
+ // a `waitFor` before this ever runs, so a failure here means the
173
+ // `waitFor` *passed* on a screen that was not the destination — and the
174
+ // labels are the only thing that can say what that screen was.
175
+ const labels = (id.entry?.targets ?? [])
176
+ .map((t) => t.label).filter(Boolean).slice(0, 12).map((l) => l.slice(0, 24));
167
177
  arrivalFailures.push(
168
- `${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`);
178
+ `${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
179
+ + `\n sensors: ${(id.entry?.sources ?? []).join('+') || 'none'}`
180
+ + `, settled: ${id.settled}, frame ${Math.round(Date.now() - (id.state?.capturedAt ?? Date.now()))}ms old`
181
+ + `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`);
169
182
  }
170
183
  }
171
184
  readings.push({
@@ -216,6 +229,12 @@ if (arrivalFailures.length) {
216
229
  console.error('measure the tour rather than the fingerprint. Fix the tour and re-run.');
217
230
  console.error('A `settle` step defaults to mode "stable", which returns instantly in the');
218
231
  console.error('moment before an animation begins — action steps already settle on their own.');
232
+ console.error('');
233
+ console.error('The tour asserts arrival with a `waitFor` before any reading is taken, so a');
234
+ console.error('failure here means that wait PASSED on a screen that was not the destination.');
235
+ console.error('Read the labels above before changing the tour: either the wait matched');
236
+ console.error('something it should not have, or the reading came off a frame older than the');
237
+ console.error('navigation — and those have opposite fixes.');
219
238
  process.exit(1);
220
239
  }
221
240
 
@@ -76,6 +76,8 @@ if (baseline > 0.65) {
76
76
  const STILL_MS_THRESHOLD = 3000;
77
77
  console.log(`\nthe brief every model arm gets is ${ollama.readBrief().length} characters, read from native/supervise.swift`);
78
78
 
79
+ const withAbstain = process.argv.includes('--abstain');
80
+
79
81
  const results = [];
80
82
  for (const arm of arms) {
81
83
  const judged = [];
@@ -87,6 +89,7 @@ for (const arm of arms) {
87
89
  const detail = {};
88
90
  const ruling = await supervisor.judge({
89
91
  ...r.situation,
92
+ mayAbstain: withAbstain,
90
93
  options: { supervisor: arm },
91
94
  // Wide on purpose. The shipped budget is 2.5s and a 14B will exceed it;
92
95
  // capping here would score the larger model on *latency* while calling it
@@ -96,7 +99,7 @@ for (const arm of arms) {
96
99
  detail,
97
100
  });
98
101
  if (!ruling) unanswered += 1;
99
- judged.push({ r, ruling, detail });
102
+ judged.push({ r, ruling, detail, abstained: detail.kind === 'abstained' });
100
103
  process.stdout.write('.');
101
104
  }
102
105
  const answered = judged.filter((j) => j.ruling);
@@ -104,6 +107,7 @@ for (const arm of arms) {
104
107
  const lat = answered.map((j) => j.ruling.ms).filter(Number.isFinite).sort((a, b) => a - b);
105
108
  results.push({
106
109
  arm,
110
+ judged,
107
111
  n: rows.length,
108
112
  unanswered,
109
113
  // Scored over every situation, not only the answered ones. A judge that
@@ -131,6 +135,24 @@ for (const r of results) {
131
135
  console.log(`${r.arm.padEnd(26)} ${pct(r.accuracy).padStart(9)} ${`${r.medianMs ?? '—'}ms`.padStart(9)} ${`${r.unanswered}`.padStart(10)}`);
132
136
  }
133
137
 
138
+ if (withAbstain) {
139
+ console.log(`\n--- item 100: the fourth word ---`);
140
+ console.log('The question is not whether it abstains. It is whether it abstains on');
141
+ console.log('the ones it would have got WRONG, or at random. A judge that declines');
142
+ console.log('uniformly has added latency and a round trip and bought nothing.\n');
143
+ console.log(`${'arm'.padEnd(22)} ${'answered'.padStart(9)} ${'of those'.padStart(9)} ${'abstained'.padStart(10)} ${'escalations'.padStart(12)}`);
144
+ console.log('-'.repeat(66));
145
+ for (const r of results) {
146
+ const answered = r.judged.filter((j) => j.ruling);
147
+ const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
148
+ const abstained = r.judged.filter((j) => j.abstained).length;
149
+ console.log(`${r.arm.padEnd(22)} ${`${answered.length}/${rows.length}`.padStart(9)} ${pct(right / (answered.length || 1)).padStart(9)} ${`${abstained}`.padStart(10)} ${pct(abstained / rows.length).padStart(12)}`);
150
+ }
151
+ console.log('\n`of those` is accuracy on the questions it chose to answer. If the fourth');
152
+ console.log('word is working, that number is higher than the three-word accuracy above');
153
+ console.log('by more than the abstention rate would give by chance.');
154
+ }
155
+
134
156
  console.log('\nerrors, by arm:');
135
157
  for (const r of results) {
136
158
  console.log(` ${r.arm}`);
@@ -140,6 +162,43 @@ for (const r of results) {
140
162
  for (const [e, n] of [...seen].sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${e}`);
141
163
  }
142
164
 
165
+ // --- the cascade: threshold -> local model -> Claude -----------------------
166
+ //
167
+ // The owner's proposal, and the numbers are the only way to say whether it
168
+ // helps: answer with the free rule where it is confident, fall through to the
169
+ // on-device model where it is not, and only then pay a round trip.
170
+ //
171
+ // **A cascade needs a tier that can decline, and neither tier has one.** The
172
+ // threshold is a comparison — it always answers. The supervisor's vocabulary is
173
+ // three words and none of them is "I don't know". So the interesting number is
174
+ // not "does a cascade help" but "how much would an abstain token be worth", and
175
+ // that is item 100. This measures it directly, by letting the rule abstain in a
176
+ // band around its own threshold and handing those to the next tier.
177
+ const armDecisions = new Map(results.map((r) => [r.arm, r.judged]));
178
+ console.log('\n--- cascade: the rule answers, the model covers where it abstains ---');
179
+ console.log(`${'abstain band'.padEnd(22)} ${'rule'.padStart(6)} ${'->model'.padStart(8)} ${'cascade'.padStart(9)} ${'model calls'.padStart(12)}`);
180
+ for (const band of [0, 250, 500, 750, 1000, 1500]) {
181
+ const lo = STILL_MS_THRESHOLD - band;
182
+ const hi = STILL_MS_THRESHOLD + band;
183
+ for (const arm of arms) {
184
+ const judged = armDecisions.get(arm) ?? [];
185
+ let right = 0;
186
+ let escalated = 0;
187
+ for (const [i, r] of rows.entries()) {
188
+ const still = r.situation.stillMs ?? 0;
189
+ if (band > 0 && still >= lo && still <= hi) {
190
+ escalated += 1;
191
+ const ruling = judged[i]?.ruling;
192
+ if (ruling && satisfies(ruling.decision, want(r))) right += 1;
193
+ continue;
194
+ }
195
+ if (satisfies(still > STILL_MS_THRESHOLD ? 'stop' : 'wait', want(r))) right += 1;
196
+ }
197
+ console.log(`${`+/-${band}ms -> ${arm}`.padEnd(22)} ${pct(stillRule).padStart(6)} ${`${escalated}`.padStart(8)} ${pct(right / rows.length).padStart(9)} ${`${escalated}/${rows.length}`.padStart(12)}`);
198
+ }
199
+ if (band === 0) console.log(' (band 0 = no abstention, the rule alone — every row below adds a tier)');
200
+ }
201
+
143
202
  console.log('\nThis scores DECISIONS on identical inputs. It cannot score outcomes —');
144
203
  console.log('whether acting on a ruling recovered the flow is a fact about the device at');
145
204
  console.log('that moment, and belongs to whichever arm was live. See docs/EXPERIMENTS.md.');
package/src/actions.js CHANGED
@@ -819,6 +819,22 @@ export async function runScript(
819
819
  fingerprint: beforeScreen?.hash ?? null,
820
820
  reason: 'verification_failed',
821
821
  candidates: [],
822
+ // Assumed, not read. A verdict says the step did not do what was
823
+ // expected; it does not say which faculty would have prevented that.
824
+ //
825
+ // Measured in the field and it matters: on an app whose controls are
826
+ // largely unlabeled, `no-visible-change` came overwhelmingly from
827
+ // tapping an inert text label whose real hit target was an invisible
828
+ // chevron — icon semantics, Phase 15 — while this line filed every
829
+ // one as evidence about sense of time, Phase 11. The tester reached
830
+ // the right conclusion from their own notes and our instrument
831
+ // disagreed with them. It was wrong.
832
+ //
833
+ // Not reclassified here, because `no-visible-change` also covers a
834
+ // switch moving 0.1% of the screen, which is neither faculty. Saying
835
+ // "assumed" is the honest answer; guessing a better-sounding reason
836
+ // would be the same mistake in the other direction.
837
+ classified: false,
822
838
  // `verification_failed` is the largest reason class in the log and it
823
839
  // was the only one carrying no intent, which made most of the corpus
824
840
  // useless for asking what kind of decision costs us. The step knows
@@ -854,6 +870,7 @@ export async function runScript(
854
870
  // Carried from the throw site where it exists, and otherwise the step's
855
871
  // own target — which is what was asked for either way.
856
872
  intent: why.intent ?? goalOf(step),
873
+ classified: why.classified,
857
874
  outcome: 'failed',
858
875
  wallMs: Date.now() - stepStart,
859
876
  detail: err.message,
@@ -2369,8 +2386,26 @@ async function runStep(deviceQuery, udid, step, ctx) {
2369
2386
  timeoutMs: step.timeoutMs ?? 8000,
2370
2387
  options: ctx.options,
2371
2388
  });
2372
- if (!w.satisfied) throw new Error(w.stalled ? w.live.note : `screen did not settle within ${w.waitedMs}ms`);
2373
- return `settled after ${w.waitedMs}ms`;
2389
+ // Say where it is still moving — item 123.
2390
+ //
2391
+ // "did not settle within 25000ms" is true and unactionable, and a failed
2392
+ // step aborts the rest of the batch, so a spinner nobody cares about can
2393
+ // cost five steps that would have worked. The reporter who asked for this
2394
+ // had landed on `pause` plus `continueOnError` and called it *"strictly
2395
+ // worse than a settle that knows what to ignore"*. Knowing what to ignore
2396
+ // is the expensive half and is not built. Saying where is nearly free —
2397
+ // the signatures were already read to detect small changes — and it is
2398
+ // the difference between re-planning a flow and ignoring a corner of it.
2399
+ if (!w.satisfied) {
2400
+ if (w.stalled) throw new Error(w.live.note);
2401
+ throw new Error(`screen did not settle within ${w.waitedMs}ms — ${settleEvidence(w)}`);
2402
+ }
2403
+ // A settle that satisfied while part of the screen is still moving says
2404
+ // so. The daemon has boxed the moving part all along and nothing read it;
2405
+ // measured at 79s of claimed stillness on a screen animating at 85ms a
2406
+ // frame. Not treated as a failure — see the note on `animatingNow`.
2407
+ return `settled after ${w.waitedMs}ms`
2408
+ + (w.animating ? ` (a ${w.animating.width}x${w.animating.height} region is still animating)` : '');
2374
2409
  }
2375
2410
  case 'waitText': {
2376
2411
  const target = step.value ?? step.text;
@@ -2385,10 +2420,12 @@ async function runStep(deviceQuery, udid, step, ctx) {
2385
2420
  // Same rule as `waitFor`, and `matchElement` says it in its own
2386
2421
  // words: a query that matched several elements has found them all
2387
2422
  // already.
2388
- if (/matched \d+ elements/.test(err.message)) {
2389
- throw new Error(
2390
- `${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
2391
- );
2423
+ // Satisfies the wait, for the same reason as `waitFor` above: several
2424
+ // matches is an answer of "yes, it is here".
2425
+ const many = err.message.match(/matched (\d+) elements/);
2426
+ if (many) {
2427
+ return `"${target}" is on screen (${many[1]} matches)`
2428
+ + ' — the wait is satisfied; pass an index to act on one of them';
2392
2429
  }
2393
2430
  }
2394
2431
  await sleep(250);
@@ -2553,10 +2590,22 @@ async function runStep(deviceQuery, udid, step, ctx) {
2553
2590
  //
2554
2591
  // Distinguished by the tag at the throw site rather than by reading
2555
2592
  // the message, because "not on this screen" tags the same reason.
2593
+ //
2594
+ // And it **satisfies** the wait rather than failing it, which is the
2595
+ // correction. The paragraph above had the reasoning right — ambiguous
2596
+ // means the target is present, several times over — and then threw
2597
+ // anyway. A `waitFor` asks one question, *has it arrived*, and two
2598
+ // matches is a yes. Reported from the field: the batch was abandoned
2599
+ // and three queued steps discarded on a screen that was exactly where
2600
+ // the flow wanted to be, and the tester had to re-issue the lot with
2601
+ // an index. The ambiguity is real and belongs in the next step's
2602
+ // selector, not in this step's verdict.
2556
2603
  if (metrics.escalationOf(err)?.ambiguous) {
2557
- throw new Error(
2558
- `${err.message}\n (not waiting: it is already on screen, and waiting cannot make it unique)`,
2559
- );
2604
+ // The count comes from the message because the tag is a boolean —
2605
+ // it marks *which kind* of ambiguity, not how many.
2606
+ const n = err.message.match(/matches (\d+) things/)?.[1];
2607
+ return `${query} is on screen${n ? ` (${n} matches)` : ' more than once'}`
2608
+ + ' — the wait is satisfied; pass an index or a #ref to act on one of them';
2560
2609
  }
2561
2610
  }
2562
2611
  if (Date.now() >= limit) break;
@@ -2657,3 +2706,41 @@ async function runStep(deviceQuery, udid, step, ctx) {
2657
2706
  throw new Error(`unknown step "${step.action}"`);
2658
2707
  }
2659
2708
  }
2709
+
2710
+ /**
2711
+ * Why a settle gave up, in the numbers that decided it.
2712
+ *
2713
+ * `screen did not settle within 25008ms` names the one quantity that is never
2714
+ * the reason, and two CI cycles went into guessing what was behind it. A wait
2715
+ * turns on three things and the message carried none of them:
2716
+ *
2717
+ * - how long the screen had actually been still when we gave up
2718
+ * - how much stillness was being asked for
2719
+ * - whether any frame arrived during the call at all
2720
+ *
2721
+ * *"Still for 1,200 ms of the 1,400 required"* and *"still for 135,000 ms and no
2722
+ * frame ever arrived"* are opposite diagnoses — the first is a budget a hair too
2723
+ * tight, the second is a stalled pipeline wearing a calm screen — and they had
2724
+ * one sentence between them. So did *"something is animating in the top right"*.
2725
+ *
2726
+ * Ordered by what to do next: what is moving, then how close stillness got,
2727
+ * then whether anything was observed, then black frames.
2728
+ */
2729
+ export function settleEvidence(w) {
2730
+ const parts = [];
2731
+ const m = w.motion;
2732
+ if (m) parts.push(`the movement is ${m.where}${m.localised ? ` (${m.share}% of it)` : ''}`);
2733
+ if (Number.isFinite(w.stillForMs) && Number.isFinite(w.stableMsRequired)) {
2734
+ parts.push(`still for ${w.stillForMs}ms of the ${w.stableMsRequired}ms required`);
2735
+ }
2736
+ // Zero frames is the loud one: a wait that observed nothing has not measured
2737
+ // this screen, it has measured a file on disk.
2738
+ if (Number.isFinite(w.framesSeen)) {
2739
+ parts.push(w.framesSeen > 0
2740
+ ? `${w.framesSeen} frame(s) arrived while waiting`
2741
+ : 'NO frame arrived while waiting — capture, not the screen');
2742
+ }
2743
+ if (w.blackFrames > 0) parts.push(`${w.blackFrames} black frame(s) — see the capture wedge`);
2744
+ return (parts.length ? parts.join('; ') : 'nothing observed')
2745
+ + (m ? `\n${m.map}` : '');
2746
+ }
package/src/analyze.js CHANGED
@@ -65,6 +65,25 @@ export function signatureDiff(a, b) {
65
65
  */
66
66
  export const CELL_CHANGE = 0.012;
67
67
 
68
+ /**
69
+ * How far two *capture paths* may differ and still be looking at one screen.
70
+ *
71
+ * Not the change threshold, and the distinction matters. `signatureDiff > 0.004`
72
+ * asks whether a screen moved between two frames from the *same* path. This asks
73
+ * whether the daemon's frame and an independent `simctl` screenshot show the
74
+ * same thing — and they never match closely, because one is downscaled by the
75
+ * capture loop and the other is a full-resolution PNG scaled here.
76
+ *
77
+ * Measured on this device, which is the only reason a number appears:
78
+ *
79
+ * same screen, two paths 0.00123 (three runs, identical to five places)
80
+ * two different screens 0.68603 (before and after a home press)
81
+ *
82
+ * A separation of 550x, so the threshold is not delicate. 0.02 is sixteen times
83
+ * the scaling cost and thirty-four times under the signal.
84
+ */
85
+ export const PATHS_AGREE = 0.02;
86
+
68
87
  /** The largest single-region change between two signatures, 0-1. */
69
88
  export function maxCellDelta(a, b) {
70
89
  const deltas = regionDeltas(a, b);
@@ -100,6 +119,43 @@ export function regionMap(deltas, cols = REGION_COLS) {
100
119
  return lines.join('\n');
101
120
  }
102
121
 
122
+ /**
123
+ * Where the movement is, in words — item 123.
124
+ *
125
+ * A settle that gives up says only that it gave up, and the reporter who asked
126
+ * for this put the cost plainly: a streaming summary panel, a Lottie and a
127
+ * support widget never settle, so the settle fails, and because a failed step
128
+ * aborts the batch it takes the other five steps with it. Their workaround was
129
+ * `pause` plus `continueOnError`, which they called *"strictly worse than a
130
+ * settle that knows what to ignore"*. Knowing what to ignore is the expensive
131
+ * half. Saying **where** is nearly free, and it is what turns "did not settle"
132
+ * into "there is a spinner in the top-right and nobody cares about it".
133
+ *
134
+ * Deliberately coarse. Thirds of the screen in each axis, named the way a person
135
+ * would point at it, and a share so a caller can tell one spinner from a screen
136
+ * that is genuinely still in flight.
137
+ */
138
+ export function describeMotion(deltas, cols = REGION_COLS) {
139
+ const total = deltas.reduce((a, b) => a + b, 0);
140
+ if (!deltas.length || total <= 0) return null;
141
+ const rows = Math.ceil(deltas.length / cols);
142
+ const bands = new Map();
143
+ const third = (i, n) => (i < n / 3 ? 0 : i < (2 * n) / 3 ? 1 : 2);
144
+ const DOWN = ['top', 'middle', 'bottom'];
145
+ const ACROSS = ['left', 'centre', 'right'];
146
+ deltas.forEach((d, i) => {
147
+ const key = `${DOWN[third(Math.floor(i / cols), rows)]}-${ACROSS[third(i % cols, cols)]}`;
148
+ bands.set(key, (bands.get(key) ?? 0) + d);
149
+ });
150
+ const ranked = [...bands.entries()].sort((a, b) => b[1] - a[1]);
151
+ const [where, amount] = ranked[0];
152
+ const share = Math.round((amount / total) * 100);
153
+ // A single band holding most of the movement is a localised animation; spread
154
+ // evenly, the whole screen is in flight and naming a corner would mislead.
155
+ if (share < 40) return { where: 'spread across the screen', share, localised: false };
156
+ return { where: where.replace('-', ' '), share, localised: true };
157
+ }
158
+
103
159
  /** Signatures live in state.json, so they are stored as compact hex. */
104
160
  export function signatureToHex(sig) {
105
161
  return sig.map((v) => v.toString(16).padStart(2, '0')).join('');
package/src/cli.js CHANGED
@@ -467,6 +467,77 @@ async function main() {
467
467
  }
468
468
 
469
469
  case 'frame': {
470
+ // `--fresh` captures independently of the daemon, and then says whether
471
+ // the two agree.
472
+ //
473
+ // This is the arbiter three field reports had to leave simframe to get.
474
+ // When `sim_look` served a three-hour-stale image labelled `130ms old`,
475
+ // the thing that finally settled it was `xcrun simctl io … screenshot` —
476
+ // run by hand, outside the tool, because nothing inside offered an
477
+ // independent read. Worse, the obvious candidate lies: `--engine` decides
478
+ // how to *start* a daemon, so passing `--engine=screenshot` to a read
479
+ // command returns the running daemon's cached frame. A tester compared
480
+ // the two, got byte-identical files with the same frame number, and
481
+ // reasonably concluded "the fallback engine is not an escape hatch".
482
+ //
483
+ // One command now answers the question the escape hatch was for: capture
484
+ // the screen twice by two different paths and report whether they agree.
485
+ if (flags.fresh) {
486
+ const dev = await resolveDevice(device);
487
+ const out = flags.out || path.join(process.cwd(), 'simframe.png');
488
+ await screenshot(dev.udid, out);
489
+ const png = fs.readFileSync(out);
490
+ // The daemon's own newest frame, for comparison. Absent is fine and
491
+ // interesting in itself: an independent capture that works while the
492
+ // daemon has none is exactly the wedge.
493
+ let cached = null;
494
+ try {
495
+ cached = await api.getFrame(device, { detail: flags.detail ?? 'normal', options });
496
+ } catch { /* no daemon, or it has nothing — reported below */ }
497
+ // Compared by *content*, never by bytes. The direct capture is a
498
+ // full-resolution PNG and the daemon's is downscaled, so a byte
499
+ // comparison says "different" every time — which is a confident wrong
500
+ // answer about the one question this command exists to settle. Both
501
+ // are decoded and reduced to the same region signature the change
502
+ // detector already uses, and `signatureDiff` is the same measure that
503
+ // decides whether a screen moved.
504
+ let diff = null;
505
+ if (cached) {
506
+ try {
507
+ const a = analyze.regionSignature(decodePng(png));
508
+ const b = analyze.regionSignature(decodePng(cached.png));
509
+ diff = analyze.signatureDiff(a, b);
510
+ } catch { /* an undecodable frame is reported as "could not compare" */ }
511
+ }
512
+ // The threshold the daemon itself calls a change. Below it the two
513
+ // paths are looking at the same screen.
514
+ const same = diff == null ? null : diff <= analyze.PATHS_AGREE;
515
+ emit(
516
+ flags,
517
+ {
518
+ file: out,
519
+ fresh: true,
520
+ bytes: png.length,
521
+ daemonSeq: cached?.state?.seq ?? null,
522
+ daemonAgeMs: cached?.ageMs ?? null,
523
+ agrees: same,
524
+ difference: diff == null ? null : Number(diff.toFixed(4)),
525
+ },
526
+ [
527
+ `${out} — captured directly from the device, ${png.length} bytes`,
528
+ cached
529
+ ? `the daemon's newest frame is #${cached.state.seq}, ${cached.ageMs}ms old`
530
+ + (same == null
531
+ ? ' — could not be compared (one of the two would not decode)'
532
+ : same
533
+ ? ` — the two paths agree (difference ${diff.toFixed(4)}, under the ${analyze.PATHS_AGREE} two paths may differ by)`
534
+ : ` — they DISAGREE (difference ${diff.toFixed(4)}). Two capture paths see different screens;`
535
+ + ' this file is the one that bypassed the daemon. `simframe revive` re-attaches capture.')
536
+ : 'the daemon has no frame to compare against, while a direct capture worked',
537
+ ],
538
+ );
539
+ return;
540
+ }
470
541
  const res = await api.getFrame(device, { detail: flags.detail ?? 'normal', options });
471
542
  const out = flags.out || path.join(process.cwd(), 'simframe.png');
472
543
  fs.writeFileSync(out, res.png);
@@ -491,7 +562,9 @@ async function main() {
491
562
  } else {
492
563
  const s = res.state;
493
564
  const out = [];
494
- if (!res.live.ok) out.push(`WARNING: ${res.live.note}`);
565
+ // Any note, not only a failing one: a dead surface reports `ok` with
566
+ // something important to say. See `liveness`.
567
+ if (res.live.note) out.push(`WARNING: ${res.live.note}`);
495
568
  // A cause, rather than five silent no-ops. Every tap on a stale
496
569
  // session is dispatched successfully and moves nothing.
497
570
  if (res.input?.stale) out.push(`input: stale — ${res.input.reason}`);
@@ -546,20 +619,36 @@ async function main() {
546
619
  changedBeforeWait: Boolean(res.changedBeforeWait),
547
620
  noVisibleChange: Boolean(res.noVisibleChange),
548
621
  stalled: Boolean(res.stalled),
622
+ // Where it was still moving, when it never stopped — item 123.
623
+ //
624
+ // `waitFor` has computed this since it learned to, and this payload
625
+ // is hand-built, so the field existed and no caller could see it. I
626
+ // read a null here and nearly concluded the tracking was broken; it
627
+ // was the reporting.
628
+ motion: res.motion ?? null,
629
+ animating: res.animating ?? null,
549
630
  hash: res.state?.hash,
550
631
  seq: res.state?.seq,
551
632
  },
552
633
  () => {
553
634
  if (res.satisfied) {
554
635
  return `${res.mode === 'change' ? 'changed' : 'settled'} after ${res.waitedMs}ms — frame #${res.state.seq}` +
555
- (res.changedBeforeWait ? ' (change had already happened before the call)' : '');
636
+ (res.changedBeforeWait ? ' (change had already happened before the call)' : '') +
637
+ (res.animating
638
+ ? `\nbut a ${res.animating.width}x${res.animating.height} region is still animating`
639
+ + ' — the stillness signal is a mean and cannot see it'
640
+ : '');
556
641
  }
557
642
  if (res.noVisibleChange) {
558
643
  return `no visible change after ${res.waitedMs}ms — screen stable, nothing moved (the action may have had no visible effect)`;
559
644
  }
560
645
  if (res.stalled) return `capture stalled after ${res.waitedMs}ms — ${res.live.note}`;
561
646
  return `timed out after ${res.waitedMs}ms — no ${res.mode === 'change' ? 'change' : 'settle'}` +
562
- (res.sawChange ? '' : '; if the change happened before this call, pass `--since` from `simframe mark`');
647
+ (res.sawChange ? '' : '; if the change happened before this call, pass `--since` from `simframe mark`') +
648
+ (res.motion
649
+ ? `\nthe movement is ${res.motion.where}`
650
+ + `${res.motion.localised ? ` (${res.motion.share}% of it)` : ''}:\n${res.motion.map}`
651
+ : '');
563
652
  },
564
653
  );
565
654
  process.exitCode = res.satisfied ? 0 : 1;
@@ -1106,13 +1195,31 @@ async function main() {
1106
1195
 
1107
1196
  case 'supervisions': {
1108
1197
  const dev = await resolveDevice(flags.device);
1109
- const records = metrics.readSupervisions(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
1198
+ // `--session` works here now, and did not before.
1199
+ //
1200
+ // Reported twice from the field: two different session ids returned
1201
+ // byte-identical output while `escalations --session` filtered correctly.
1202
+ // A flag that exists on one command and is silently inert on its sibling
1203
+ // is worse than an absent one — this command's own footer warns that the
1204
+ // counts pool multiple agents and then offered no way to unpool them.
1205
+ const session = flags.session === true
1206
+ ? metrics.sessionId()
1207
+ : (flags.session ? String(flags.session) : null);
1208
+ const all = metrics.readSupervisions(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
1209
+ const records = session ? all.filter((r) => r?.session_id === session) : all;
1110
1210
  const b = metrics.supervisionBreakdown(records);
1111
1211
  if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify({ ...b, records }, null, 2)}\n`);
1112
1212
  emit(flags, { ...b, records: flags.verbose ? records : undefined }, [
1113
1213
  `${b.total} supervisor ruling${b.total === 1 ? '' : 's'} on ${dev.name}`,
1114
- b.total ? '' : 'Nothing has been judged on this device yet. The supervisor is off unless'
1115
- + ' SIMFRAME_SUPERVISOR=apple, and a ruling is only recorded when a step actually fails.',
1214
+ // "Nothing here" and "nothing matched your filter" are different
1215
+ // answers, and the first one told a reader the supervisor had never
1216
+ // run on a device holding 111 rulings.
1217
+ b.total
1218
+ ? null
1219
+ : (all.length
1220
+ ? `no ruling in this log belongs to session ${session} — the device has ${all.length}.`
1221
+ : 'Nothing has been judged on this device yet. The supervisor is off unless'
1222
+ + ' SIMFRAME_SUPERVISOR=apple, and a ruling is only recorded when a step actually fails.'),
1116
1223
  ...Object.entries(b.decision_to_outcome)
1117
1224
  .sort((a, c) => c[1] - a[1])
1118
1225
  .map(([k, n]) => ` ${k.padEnd(28)} ${String(n).padStart(4)}`),
@@ -1128,8 +1235,12 @@ async function main() {
1128
1235
  ? ` — ${b.p95_unknown} ruling(s) are on edges with no p95, so they cannot take part in 101's comparison`
1129
1236
  : '')
1130
1237
  : null,
1238
+ session && all.length !== records.length
1239
+ ? `filtered to session ${session}: ${records.length} of ${all.length} ruling(s)`
1240
+ : null,
1131
1241
  b.sessions.length > 1
1132
1242
  ? `WARNING ${b.sessions.length} sessions are pooled here; two agents on one device write one file`
1243
+ + ' — narrow with --session (this process) or --session=<id>'
1133
1244
  : null,
1134
1245
  ].filter((l) => l !== null).join('\n'));
1135
1246
  break;
@@ -1148,10 +1259,29 @@ async function main() {
1148
1259
  ...metrics.REASONS
1149
1260
  .filter((r) => b.by_reason[r])
1150
1261
  .sort((a, c) => b.by_reason[c] - b.by_reason[a])
1151
- .map((r) => ` ${r.padEnd(20)} ${String(b.by_reason[r]).padStart(4)} `
1152
- + (metrics.BUILT_FACULTIES.has(metrics.FACULTY[r])
1262
+ .map((r) => {
1263
+ const n = b.by_reason[r];
1264
+ const read = b.classified_by_reason?.[r] ?? 0;
1265
+ // A faculty is only named for the part of a reason that was read
1266
+ // off the failure. The rest is a count of things nothing could
1267
+ // classify, and naming a phase against it is advice with nothing
1268
+ // behind it — which is how this report came to tell a tester that
1269
+ // their unlabeled-control problem was a timing problem.
1270
+ const assumed = b.assumed_by_reason?.[r] ?? 0;
1271
+ const named = metrics.BUILT_FACULTIES.has(metrics.FACULTY[r])
1153
1272
  ? `not removed by: ${metrics.FACULTY[r]} [built]`
1154
- : `would be removed by: ${metrics.FACULTY[r]}`)),
1273
+ : `would be removed by: ${metrics.FACULTY[r]}`;
1274
+ let verdict;
1275
+ if (read > 0) {
1276
+ verdict = named + (read < n ? ` (on the ${read} of ${n} whose reason was read)` : '');
1277
+ } else if (assumed > 0) {
1278
+ verdict = 'reason assumed, not read — no faculty can be named from these';
1279
+ } else {
1280
+ // Neither read nor assumed: the log predates the distinction.
1281
+ verdict = `${named} — but these records predate the check, so treat it as untested`;
1282
+ }
1283
+ return ` ${r.padEnd(20)} ${String(n).padStart(4)} ${verdict}`;
1284
+ }),
1155
1285
  b.total ? '' : null,
1156
1286
  b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
1157
1287
  // Said out loud rather than left for someone to discover: the rate is
@@ -1525,9 +1655,38 @@ async function doctor({ json = false, strict = false, device, options = {} } = {
1525
1655
  // machine could be doing better and silently is not; a driver someone
1526
1656
  // selected on purpose is neither silent nor a surprise.
1527
1657
  const axState = !ax.available ? 'optional' : ax.name === 'simframed' || ax.chosen ? 'ok' : 'warn';
1528
- add(`accessibility tree (${d.name})`, axState,
1529
- ax.available ? `${ax.name}: ${ax.version}` : `unavailable: ${ax.reason}`,
1658
+ // Prove a round trip, not a presence — the same correction this file
1659
+ // already made for the supervisor, never carried across to here.
1660
+ //
1661
+ // A CI run read the screen eighteen times and every single reading came
1662
+ // back `ocr` with no `ax` at all, while this check said `ok` because a
1663
+ // driver was configured. It is configured; it answers with nothing. Six
1664
+ // minutes later the fingerprint step failed with a distribution mystery,
1665
+ // and the layer that had actually died was named nowhere. Asking the tree
1666
+ // for the current screen costs one read (~50ms) and turns that into a
1667
+ // first-minute failure with the right sentence on it.
1668
+ let axCount = null;
1669
+ if (ax.available) {
1670
+ try {
1671
+ axCount = (await input.describeAll(d.udid)).length;
1672
+ } catch {
1673
+ axCount = 0;
1674
+ }
1675
+ }
1676
+ add(`accessibility tree (${d.name})`,
1677
+ // `warn`, not `fail`: a genuinely empty screen exists — a springboard
1678
+ // mid-boot, a black frame — and a hard error on one would cry wolf.
1679
+ // The count is exported so a caller that knows the screen is not empty
1680
+ // can assert on it, which is what CI does.
1681
+ ax.available && axCount === 0 ? 'warn' : axState,
1682
+ ax.available
1683
+ ? `${ax.name}: ${ax.version}`
1684
+ + (axCount === 0
1685
+ ? ' — but it returned NO elements for the current screen, so every read is OCR alone'
1686
+ : axCount != null ? `; ${axCount} element(s) on the current screen` : '')
1687
+ : `unavailable: ${ax.reason}`,
1530
1688
  { key: 'ax.driver', value: ax.name });
1689
+ if (axCount != null) add(null, null, null, { key: 'ax.elements', value: axCount });
1531
1690
  }
1532
1691
  if (probed.length) {
1533
1692
  const t0 = Date.now();
@@ -1554,6 +1713,13 @@ async function doctor({ json = false, strict = false, device, options = {} } = {
1554
1713
  for (const d of probed) {
1555
1714
  const live = api.liveness(d.udid, (await api.getState(d.udid)).state);
1556
1715
  if (live.stalled) add(`capture health (${d.name})`, 'fail', live.note, { key: 'capture.stalled', value: true });
1716
+ // A `warn` rather than a `fail`, because a genuinely inert screen is
1717
+ // possible and this is a contradiction between two numbers rather than
1718
+ // a proven fault. It is still the loudest thing `doctor` can say about
1719
+ // the failure that made a tester report a false application state.
1720
+ else if (live.suspectSurface) {
1721
+ add(`capture surface (${d.name})`, 'warn', live.note, { key: 'capture.suspectSurface', value: true });
1722
+ }
1557
1723
  }
1558
1724
  }
1559
1725
  } catch (err) {
@@ -1594,11 +1760,14 @@ async function doctor({ json = false, strict = false, device, options = {} } = {
1594
1760
  warnings: warned.length,
1595
1761
  optional: optional.length,
1596
1762
  ...flat,
1597
- checks: checks.map(({ name, level, detail }) => ({ name, level, detail })),
1763
+ checks: checks.filter((c) => c.name).map(({ name, level, detail }) => ({ name, level, detail })),
1598
1764
  }, null, 2));
1599
1765
  } else {
1600
1766
  const mark = { ok: 'ok ', warn: 'WARN', fail: 'FAIL', optional: '-- ' };
1601
- for (const c of checks) console.log(`${mark[c.level]} ${c.name.padEnd(24)} ${c.detail}`);
1767
+ // A nameless entry is data for `--json` and not a line for a reader — the
1768
+ // element count belongs beside the layer it describes, not on a row of its
1769
+ // own.
1770
+ for (const c of checks) if (c.name) console.log(`${mark[c.level]} ${c.name.padEnd(24)} ${c.detail}`);
1602
1771
  if (warned.length) {
1603
1772
  console.log(`\n${warned.length} layer(s) degraded. simframe still works, but not at full speed or coverage:`);
1604
1773
  for (const c of warned) console.log(` - ${c.name}: ${c.detail}`);
@@ -72,8 +72,17 @@ import * as regions from './regions.js';
72
72
  * graph merged them. Both bounds are absolute now. Screens that were
73
73
  * missed at 7 hash differently at 8, and unlike a stale hash that matches
74
74
  * nothing, these matched the *wrong* thing.
75
+ *
76
+ * 9 — nothing in this file changed. The element list it is given did: item 122
77
+ * stopped dropping accessibility nodes that have no name, so a screen with
78
+ * an icon-only control now carries a token for it that it did not carry
79
+ * before. That is a better identity — a nav bar with an overflow menu and
80
+ * one without are not the same screen — and it is still a different hash
81
+ * for the same screen, which is what this number exists to declare. The
82
+ * lesson worth keeping is that the rules version is not a version of *this
83
+ * file*; it is a version of the token set, and the token set has an input.
75
84
  */
76
- export const TOKEN_RULES_VERSION = 8;
85
+ export const TOKEN_RULES_VERSION = 9;
77
86
 
78
87
  /** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
79
88
  export const GRID = 24;