simframe 0.14.2 → 0.14.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "simframe",
3
- "version": "0.14.2",
3
+ "version": "0.14.3",
4
4
  "mcpName": "io.github.lvlrSajjad/simframe",
5
5
  "description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
6
6
  "keywords": [
@@ -0,0 +1,121 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * Does the version on npm contain the code you think it does?
4
+ *
5
+ * **DEFERRED 149.** `0.14.1` was tagged at one commit and a change landed
6
+ * immediately after it, so `npm install simframe@0.14.1` shipped the previous
7
+ * `centerOf()`. An external tester was asked to evaluate that change, diffed
8
+ * the package against the checkout themselves, and wrote: *"I came within one
9
+ * command of filing this report against the wrong binary."*
10
+ *
11
+ * Nothing in the pipeline compared what was tagged against what was being asked
12
+ * for, and the release workflow cannot: it checks that the tag, `package.json`
13
+ * and `server.json` agree, which they did. The gap is between the tag and
14
+ * whatever arrived afterwards.
15
+ *
16
+ * So this is the check to run **before handing someone a version to test**, and
17
+ * before cutting the next one. It fetches the published tarball and compares
18
+ * every shipped source file against the working tree.
19
+ *
20
+ * It deliberately compares against the WORKING TREE rather than a tag: the
21
+ * question being asked is "is what I am about to ask someone to install the code
22
+ * I am looking at", and a tag cannot answer that.
23
+ *
24
+ * Usage:
25
+ * node scripts/check-published.mjs # the version in package.json
26
+ * node scripts/check-published.mjs 0.14.1 # any published version
27
+ * node scripts/check-published.mjs --latest # whatever npm serves as latest
28
+ *
29
+ * Exit 0 when every shipped file matches, 1 when any differs, 2 when the
30
+ * version is not published or npm could not be reached — which is a different
31
+ * answer and must not read as "it matches".
32
+ */
33
+ import { execFileSync } from 'node:child_process';
34
+ import { createHash } from 'node:crypto';
35
+ import fs from 'node:fs';
36
+ import os from 'node:os';
37
+ import path from 'node:path';
38
+
39
+ const md5 = (buf) => createHash('md5').update(buf).digest('hex');
40
+ const here = path.resolve(path.dirname(new URL(import.meta.url).pathname), '..');
41
+
42
+ const arg = process.argv.slice(2).find((a) => !a.startsWith('-'));
43
+ const wantLatest = process.argv.includes('--latest');
44
+ const local = JSON.parse(fs.readFileSync(path.join(here, 'package.json'), 'utf8'));
45
+
46
+ let version = arg ?? local.version;
47
+ if (wantLatest) {
48
+ try {
49
+ version = execFileSync('npm', ['view', local.name, 'version'], { encoding: 'utf8' }).trim();
50
+ } catch (err) {
51
+ console.error(`could not ask npm for the latest ${local.name}: ${err.message}`);
52
+ process.exit(2);
53
+ }
54
+ }
55
+
56
+ console.log(`comparing ${local.name}@${version} on npm against this working tree`);
57
+
58
+ const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'simframe-published-'));
59
+ let tarball;
60
+ try {
61
+ tarball = execFileSync('npm', ['pack', `${local.name}@${version}`, '--silent', '--pack-destination', tmp], {
62
+ encoding: 'utf8',
63
+ }).trim().split('\n').pop();
64
+ } catch (err) {
65
+ // Not published, unpublished, or no network. None of those is a match.
66
+ console.error(`could not fetch ${local.name}@${version} from npm.`);
67
+ console.error('That is not the same answer as "it differs" — nothing was compared.');
68
+ console.error(String(err.stderr || err.message).trim().split('\n').slice(-3).join('\n'));
69
+ process.exit(2);
70
+ }
71
+
72
+ execFileSync('tar', ['-xzf', path.join(tmp, tarball), '-C', tmp]);
73
+ const root = path.join(tmp, 'package');
74
+
75
+ /** Every file the package ships, relative to the package root. */
76
+ const shipped = [];
77
+ const walk = (dir) => {
78
+ for (const e of fs.readdirSync(dir, { withFileTypes: true })) {
79
+ const full = path.join(dir, e.name);
80
+ if (e.isDirectory()) walk(full);
81
+ else shipped.push(path.relative(root, full));
82
+ }
83
+ };
84
+ walk(root);
85
+
86
+ // Only compare what this repo is the source of. `package.json` is rewritten by
87
+ // npm on publish (it adds `_id`, `dist`, and normalises fields), so byte
88
+ // equality there is not the question and would be a permanent false alarm.
89
+ const comparable = shipped.filter((f) => /^(src|scripts|native)\//.test(f) && !f.includes('/.build/'));
90
+
91
+ const same = [];
92
+ const differ = [];
93
+ const missing = [];
94
+ for (const rel of comparable) {
95
+ const mine = path.join(here, rel);
96
+ if (!fs.existsSync(mine)) { missing.push(rel); continue; }
97
+ if (md5(fs.readFileSync(mine)) === md5(fs.readFileSync(path.join(root, rel)))) same.push(rel);
98
+ else differ.push(rel);
99
+ }
100
+
101
+ const publishedVersion = JSON.parse(fs.readFileSync(path.join(root, 'package.json'), 'utf8')).version;
102
+ console.log(` published version: ${publishedVersion}`);
103
+ console.log(` files compared: ${comparable.length}`);
104
+ console.log(` identical: ${same.length}`);
105
+
106
+ if (missing.length) {
107
+ console.log(` shipped but absent here: ${missing.length}`);
108
+ for (const f of missing) console.log(` ${f}`);
109
+ }
110
+
111
+ if (!differ.length && !missing.length) {
112
+ console.log(`\nok — ${local.name}@${version} is the code in this tree`);
113
+ process.exit(0);
114
+ }
115
+
116
+ console.error(`\nFAIL ${differ.length} shipped file(s) differ from this tree:`);
117
+ for (const f of differ) console.error(` ${f}`);
118
+ console.error('\nIf you are about to ask someone to test a change, they would be testing');
119
+ console.error('the published bytes above, not what you are reading. Publish first, or point');
120
+ console.error('them at the checkout and say so.');
121
+ process.exit(1);
@@ -31,6 +31,24 @@ const DEVICE_STATE = [
31
31
  [/Timeout waiting for screen surfaces|display surface is not answering|display surface could not be read/i, 'the display surface is wedged'],
32
32
  [/no frames buffered|capture is wedged/i, 'capture stopped'],
33
33
  [/the second app never launched|could not be dispatched/i, 'an app would not launch'],
34
+ // A launched app that never comes to the front, seen as the tour waiting for
35
+ // one of its landmarks on a screen that is showing a clock and nothing else.
36
+ //
37
+ // Measured on a runner: `ok launch — launched com.apple.Preferences
38
+ // (relaunched)` followed by `waited 8000ms for General: "General" is not on
39
+ // this screen. Visible: 10:50, .?o (the screen has not moved for 6181ms)`.
40
+ // Two labels, one of them a clock, on a still screen — the device is not
41
+ // presenting the app, and the guard called that a check failing on its
42
+ // merits and declined to revive.
43
+ //
44
+ // Deliberately narrow. It requires the wait to have failed AND the screen to
45
+ // have been still AND almost nothing readable: a tour that genuinely asks for
46
+ // the wrong label has a screen full of other labels, and must keep failing
47
+ // rather than being retried into a pass.
48
+ [
49
+ /never arrived[\s\S]*?Visible:[^\n]{0,24}\(the screen has not moved for \d+ms/i,
50
+ 'a launched app never came to the front (the screen shows a clock and nothing else)',
51
+ ],
34
52
  ];
35
53
 
36
54
  const udid = process.argv[2];
@@ -0,0 +1,63 @@
1
+ /**
2
+ * Why does a fingerprint reading not resemble its own screen?
3
+ *
4
+ * Its own module, with no side effects, for one reason: this logic has failed
5
+ * at RUNTIME twice — once reaching for a variable local to another function,
6
+ * once on declaration order — while being correct both times. `eval-fingerprint.mjs`
7
+ * runs the whole eval on import, so nothing could test it there, and the only
8
+ * thing that ever exercised it was a hosted runner at the end of a
9
+ * fifteen-minute job, in the middle of a report. That is the most expensive
10
+ * place in this project to find a typo.
11
+ *
12
+ * Three causes, and they have different remedies — which is the whole reason to
13
+ * tell them apart rather than print one sentence about all three:
14
+ *
15
+ * - **collided** — neither reading carries a chrome label, so both are
16
+ * structure with no name and the fingerprint has nothing left to separate two
17
+ * list screens. That is this harness's own subject, and a real finding.
18
+ * - **wrongScreen** — the tokens are identical to another screen AND this
19
+ * screen reads differently in its other rounds, so it is demonstrably
20
+ * distinguishable and the tour was simply somewhere else. A tap that missed.
21
+ * - **underRead** — the reading stayed under the token floor and the screen
22
+ * cannot tell itself apart in any round, so we never looked long enough. Ours
23
+ * to fix, and nothing about the tour or the fingerprint.
24
+ *
25
+ * The middle case is the correction that prompted this. Sparseness alone used
26
+ * to claim `underRead`, and a wrong turn onto a screen that *legitimately*
27
+ * reads sparse — the Settings root, at 4 tokens on a runner — is flagged sparse
28
+ * too. So a genuine tour failure was reported as our instrument's fault, and
29
+ * the harness's original and correct message had been silenced by an
30
+ * "improvement".
31
+ */
32
+ import * as fingerprint from '../src/fingerprint.js';
33
+
34
+ /**
35
+ * @param {object} o
36
+ * @param {object} o.reading the stray
37
+ * @param {object|null} o.match the other screen's reading it most resembles
38
+ * @param {number} o.bestOther how much it resembles that one, 0..1
39
+ * @param {object[]} o.siblings every reading of the stray's own screen
40
+ * @param {boolean} o.wasSparse did it stay under the token floor after retries
41
+ * @param {string[]} [o.named] chrome-labelled tokens in the stray
42
+ * @param {string[]} [o.matchNamed] chrome-labelled tokens in the match
43
+ */
44
+ export function classifyStray({
45
+ reading, match, bestOther, siblings, wasSparse, named = [], matchNamed = [],
46
+ }) {
47
+ // Does any other round of this same screen read differently from the screen
48
+ // we collided with? If so this screen CAN be told apart, and a round that
49
+ // matched the other one exactly was somewhere else.
50
+ const distinguishable = (siblings ?? [])
51
+ .some((o) => o !== reading && fingerprint.similarity(o.tokens, match?.tokens ?? []) < 0.99);
52
+ const identical = bestOther >= 0.99;
53
+ // Checked first and exclusively: a reading with no names at all cannot be
54
+ // said to have gone anywhere, because there is nothing in it that would have
55
+ // named a destination.
56
+ const collided = identical && named.length === 0 && matchNamed.length === 0;
57
+ return {
58
+ collided,
59
+ wrongScreen: !collided && identical && distinguishable,
60
+ underRead: !collided && Boolean(wasSparse) && !distinguishable,
61
+ distinguishable,
62
+ };
63
+ }
@@ -20,6 +20,7 @@
20
20
  import fs from 'node:fs';
21
21
  import * as actions from '../src/actions.js';
22
22
  import * as fingerprint from '../src/fingerprint.js';
23
+ import { classifyStray } from './classify-stray.mjs';
23
24
  import * as graph from '../src/graph.js';
24
25
  import * as api from '../src/index.js';
25
26
 
@@ -152,12 +153,24 @@ const STALE_READ_RETRIES = 3;
152
153
  * Backed off rather than fixed, because the thing being waited for is a screen
153
154
  * finishing its draw, and the runner that needs this is the slow one.
154
155
  */
156
+ /**
157
+ * How old a frame may be and still be a reading of NOW.
158
+ *
159
+ * `FRAME_IS_CURRENT_MS` in src/index.js is 2500 for the same question on the
160
+ * settle path. Doubled here because this harness deliberately reads cold and a
161
+ * hosted runner's capture p50 is 2.5x this laptop's — but bounded, because a
162
+ * 38-second-old frame is a different screen, not a slow one.
163
+ */
164
+ const FRAME_TOO_OLD_MS = 5000;
165
+
155
166
  const SPARSE_READ_RETRIES = 3;
156
167
  const SPARSE_READ_WAIT_MS = 1500;
157
168
  const STALE_READ_WAIT_MS = 1000;
158
169
 
159
170
  /** When the steps that were supposed to change the screen finished. */
160
171
  let navigatedAt = 0;
172
+ /** What the tour's own steps said, kept for the arrival report. */
173
+ let lastSteps = [];
161
174
  /** Readings that never got a frame newer than their own navigation. */
162
175
  const staleReadings = [];
163
176
 
@@ -222,7 +235,37 @@ for (let round = 1; round <= rounds; round += 1) {
222
235
  // written out first, because a failing run is the one whose evidence
223
236
  // matters.
224
237
  try {
225
- await actions.runScript(device, { steps: screen.steps, verify: false });
238
+ // Keep what the tour's own steps reported. On an arrival failure the
239
+ // decisive question is whether the last wait PASSED and on what — and
240
+ // that answer was being thrown away, so three CI failures in a row
241
+ // could only be reasoned about rather than read. A local reproduction
242
+ // ran the same sequence 14 times and landed correctly every time, so
243
+ // the runner is the only place this can be observed.
244
+ const run = await actions.runScript(device, { steps: screen.steps, verify: false });
245
+ lastSteps = (run?.results ?? []).map((r) => ({
246
+ action: r.action,
247
+ ok: r.ok !== false,
248
+ ms: r.ms,
249
+ detail: typeof r.detail === 'string' ? r.detail.slice(0, 200) : null,
250
+ }));
251
+ // **`runScript` does not throw on a failed step** — it returns
252
+ // `ok: !failed` — and this only ever caught throws. So a tour whose
253
+ // `waitFor` timed out was treated as having arrived: `navigatedAt` was
254
+ // set, the previous screen was read, and the arrival check then reported
255
+ // it as a fingerprint result. Three CI failures were that, and the
256
+ // diagnostic added one commit earlier is what showed it —
257
+ // `ok launch … / FAIL waitFor` sitting under a reading the harness had
258
+ // accepted.
259
+ //
260
+ // A tour that did not arrive is a precondition, not a measurement. It
261
+ // throws into the same handler as a step that threw, so the report is
262
+ // identical and the readings so far are still written out.
263
+ if (run && run.ok === false) {
264
+ const bad = (run.results ?? []).find((r) => r.ok === false);
265
+ throw new Error(
266
+ `step ${bad?.action ?? '?'} did not succeed: ${bad?.error ?? 'no reason given'}`,
267
+ );
268
+ }
226
269
  navigatedAt = Date.now();
227
270
  } catch (err) {
228
271
  save({ abandonedAt: { screen: screen.name, round, error: err.message } });
@@ -275,18 +318,38 @@ for (let round = 1; round <= rounds; round += 1) {
275
318
  // stale — an eval that scores a stale frame is measuring the runner, which
276
319
  // is the one thing this harness says it is not doing.
277
320
  if (navigatedAt) {
321
+ // TWO conditions, and the second was missing.
322
+ //
323
+ // Newer than the navigation is not the same as recent. Measured on a
324
+ // runner: a reading came off a frame **38,266 ms old** and passed this
325
+ // guard, because the navigation had also been more than 38 s earlier — so
326
+ // `capturedAt >= navigatedAt` held while the frame was half a minute
327
+ // stale and showed a screen two steps back. `settled: true` alongside it,
328
+ // because a starved capture looks exactly like a still one.
329
+ //
330
+ // A frame this old is capture starvation, not a slow screen, and no
331
+ // number of re-reads invents a frame the daemon is not producing. So it
332
+ // retries and then says which of the two it is, because "the frame
333
+ // predates the navigation" and "capture stopped producing frames" point
334
+ // at completely different things.
335
+ const tooOld = (r) => Date.now() - (r.state?.capturedAt ?? 0) > FRAME_TOO_OLD_MS;
336
+ const predatesNav = (r) => (r.state?.capturedAt ?? 0) < navigatedAt;
278
337
  for (let attempt = 0; attempt < STALE_READ_RETRIES; attempt += 1) {
279
- const capturedAt = id.state?.capturedAt ?? 0;
280
- if (capturedAt >= navigatedAt) break;
281
- const age = Date.now() - capturedAt;
282
- console.log(` ("${screen.name}" read a frame from ${age}ms ago, older than the navigation`
338
+ if (!predatesNav(id) && !tooOld(id)) break;
339
+ const age = Date.now() - (id.state?.capturedAt ?? 0);
340
+ console.log(` ("${screen.name}" read a frame from ${age}ms ago`
341
+ + `${predatesNav(id) ? ', older than the navigation' : `, over the ${FRAME_TOO_OLD_MS}ms freshness bound`}`
283
342
  + ` — reading again ${attempt + 1}/${STALE_READ_RETRIES})`);
284
343
  await new Promise((r) => setTimeout(r, STALE_READ_WAIT_MS));
285
344
  id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
286
345
  }
287
- if ((id.state?.capturedAt ?? 0) < navigatedAt) {
346
+ if (predatesNav(id)) {
288
347
  staleReadings.push(`${screen.name} round ${round}: frame still predates the navigation`
289
348
  + ` by ${navigatedAt - (id.state?.capturedAt ?? 0)}ms after ${STALE_READ_RETRIES} re-reads`);
349
+ } else if (tooOld(id)) {
350
+ staleReadings.push(`${screen.name} round ${round}: newest frame is`
351
+ + ` ${Date.now() - (id.state?.capturedAt ?? 0)}ms old after ${STALE_READ_RETRIES} re-reads`
352
+ + ' — capture is starved, not the screen still');
290
353
  }
291
354
  }
292
355
  // A reading too sparse to be a screen is not a reading.
@@ -343,6 +406,7 @@ for (let round = 1; round <= rounds; round += 1) {
343
406
  // taken, and every distribution below it is then measuring the tour rather
344
407
  // than the fingerprint. The first version of this harness did exactly that
345
408
  // and reported that the distributions overlapped completely.
409
+ id.steps = lastSteps;
346
410
  const previous = readings[readings.length - 1];
347
411
  if (previous && previous.name !== screen.name) {
348
412
  // Not just an identical hash. Two readings of the same screen can differ
@@ -367,10 +431,20 @@ for (let round = 1; round <= rounds; round += 1) {
367
431
  `${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
368
432
  + `\n sensors: ${(id.entry?.sources ?? []).join('+') || 'none'}`
369
433
  + `, settled: ${id.settled}, frame ${Math.round(Date.now() - (id.state?.capturedAt ?? Date.now()))}ms old`
370
- + `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`);
434
+ + `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`
435
+ // The decisive line, and it was missing. Whether the tour's last wait
436
+ // PASSED, and what it said, separates "the wait matched something it
437
+ // should not have" from "the screen changed after the wait was
438
+ // satisfied" — the two causes this message has always named as having
439
+ // opposite fixes, while giving the reader nothing to tell them apart.
440
+ + `\n the tour's own steps: ${
441
+ (lastSteps ?? []).length
442
+ ? lastSteps.map((st) => `${st.ok ? 'ok' : 'FAIL'} ${st.action}${st.detail ? ` — ${st.detail}` : ''}`).join('\n ')
443
+ : '(not recorded)'}`);
371
444
  }
372
445
  }
373
446
  readings.push({
447
+ steps: lastSteps,
374
448
  name: screen.name,
375
449
  round,
376
450
  hash: id.hash,
@@ -506,6 +580,17 @@ function findStrays(all) {
506
580
  }
507
581
 
508
582
  const strays = findStrays(readings);
583
+ // Readings grouped by screen, for the classification below.
584
+ //
585
+ // `findStrays` builds this too and it is LOCAL to that function — so reaching
586
+ // for it here threw `byName is not defined` on a runner, in the middle of the
587
+ // report, after the strays had already been printed. I had validated the
588
+ // classification against a saved artifact in a standalone script, where I had
589
+ // declared it myself, and never once in place. The logic was right and the
590
+ // integration was never run: the unit tests do not reach this file at all.
591
+ const readingsByName = new Map();
592
+ for (const r of readings) readingsByName.set(r.name, [...(readingsByName.get(r.name) ?? []), r]);
593
+
509
594
  if (strays.length) {
510
595
  console.error(`\nFAIL ${strays.length} reading(s) do not resemble their own screen:`);
511
596
  // Two causes wear the same symptom, and until now the report asserted the
@@ -516,10 +601,25 @@ if (strays.length) {
516
601
  // label while matching one particular other screen almost exactly, and a
517
602
  // wrong turn is one whose tokens name a screen the tour did not ask for.
518
603
  let collisions = 0;
604
+ // Kept so the SUMMARY agrees with the lines above it. It did not: the summary
605
+ // tested `sparseAt` directly while each line used the classifier, so a stray
606
+ // correctly labelled WRONG SCREEN was followed by "every stray above was
607
+ // UNDER-READ". A report that contradicts itself in consecutive paragraphs is
608
+ // worse than either sentence alone.
609
+ const verdicts = [];
519
610
  for (const { reading, bestSelf, bestOther, match } of strays) {
520
611
  const named = namedTokens(reading.tokens);
521
612
  const matchNamed = match ? namedTokens(match.tokens) : [];
522
- const collided = bestOther >= 0.99 && named.length === 0 && matchNamed.length === 0;
613
+ // One classifier, one source of truth. `collided` was computed here as well
614
+ // and the two could drift — which is how this block came to have a third
615
+ // cause landing in the wrong bucket in the first place.
616
+ const verdict = classifyStray({
617
+ reading, match, bestOther, named, matchNamed,
618
+ siblings: readingsByName.get(reading.name) ?? [],
619
+ wasSparse: sparseAt.has(`${reading.name}|${reading.round}`),
620
+ });
621
+ const { collided, underRead, wrongScreen } = verdict;
622
+ verdicts.push(verdict);
523
623
  if (collided) collisions += 1;
524
624
  // The third cause, and the one that produced this report on 2026-09-15.
525
625
  //
@@ -549,10 +649,6 @@ if (strays.length) {
549
649
  // distinguishable and a round matching another screen exactly went
550
650
  // somewhere else. Only when no round of this screen can tell itself apart
551
651
  // is "we did not look long enough" the honest reading.
552
- const distinguishable = (byName.get(reading.name) ?? [])
553
- .some((o) => o !== reading && fingerprint.similarity(o.tokens, match?.tokens ?? []) < 0.99);
554
- const underRead = sparseAt.has(`${reading.name}|${reading.round}`) && !distinguishable;
555
- const wrongScreen = bestOther >= 0.99 && distinguishable;
556
652
  console.error(` ${reading.name} r${reading.round}: own screen ${bestSelf.toFixed(2)}, `
557
653
  + `${match ? `${match.name} r${match.round}` : 'another screen'} ${bestOther.toFixed(2)} `
558
654
  + `(${reading.count} tokens, ${named.length} named, sources ${reading.sources.join('+') || 'none'})`);
@@ -576,10 +672,16 @@ if (strays.length) {
576
672
  console.error(`\n${collisions} of ${strays.length} are fingerprint collisions. That is this harness's own subject,`);
577
673
  console.error('not a tour fault: a reading whose chrome label went missing cannot establish');
578
674
  console.error('identity, and comparing it as though it could is what produced the verdict above.');
579
- } else if (strays.every((x) => sparseAt.has(`${x.reading.name}|${x.reading.round}`))) {
675
+ } else if (verdicts.length && verdicts.every((v) => v.underRead)) {
580
676
  console.error('\nEvery stray above was UNDER-READ, so this says nothing about the tour or the');
581
677
  console.error('fingerprint — the harness scored a look that was too short. The retries are in');
582
678
  console.error('SPARSE_READ_RETRIES; a runner that needs more than they allow is the finding.');
679
+ } else if (verdicts.some((v) => v.wrongScreen)) {
680
+ const n = verdicts.filter((v) => v.wrongScreen).length;
681
+ console.error(`\n${n} stray(s) were read on the WRONG SCREEN — the tokens match another screen`);
682
+ console.error('exactly while this one reads differently in its other rounds, so it is');
683
+ console.error('distinguishable and the tour was somewhere else. Not a fingerprint result:');
684
+ console.error('a tap that missed, or a screen that went back before it was read.');
583
685
  } else {
584
686
  console.error('\nThat is the tour going somewhere unintended, not the fingerprint drifting, and');
585
687
  console.error('measuring it as either distribution poisons both ends. Fix the tour — a tap that');
package/src/actions.js CHANGED
@@ -732,7 +732,10 @@ export async function runScript(
732
732
  // eight times below its threshold, so it reads as no-visible-change and
733
733
  // its edge is no longer recorded. That is the right trade while the
734
734
  // detector cannot see it, and it comes back on its own once it can.
735
- const noEvidence = Boolean(settled?.noVisibleChange);
735
+ // A changed control IS evidence, whatever the frame hash says. Without
736
+ // this the edge is not recorded either, so a toggle that worked teaches
737
+ // the graph nothing and stays unpredictable forever.
738
+ const noEvidence = Boolean(settled?.noVisibleChange) && !stateDelta(beforeScreen?.entry, afterReading?.entry);
736
739
  if (afterScreen.confirmed && afterScreen.hash && !noEvidence) {
737
740
  // The observed cost of this transition, which is what makes the next
738
741
  // one adaptive. Only from a settle that was actually satisfied: a
@@ -788,7 +791,22 @@ export async function runScript(
788
791
  // first — and six swipes reporting "no visible change" is what item 120
789
792
  // was reported against. Keying on one of them would have shipped a fix
790
793
  // that did not fire on its own bug report.
791
- const wentNowhere = Boolean(settled?.noVisibleChange) || verification?.verdict === 'no-visible-change';
794
+ // Declared BEFORE `wentNowhere`, which reads it. It sat below and the
795
+ // unit tests were green because nothing without a device reaches this
796
+ // line — a `const` in the temporal dead zone throws on first use, which
797
+ // is the same trap the `afterReading` comment above records.
798
+ //
799
+ // Computed on EITHER signal, because they are two different sensors and
800
+ // the reported case came through the one this first version missed: the
801
+ // radio move settled in 2714ms with `sawChange` true, so
802
+ // `settled.noVisibleChange` was false and the verdict came from the
803
+ // fingerprint instead.
804
+ const changedState = (settled?.noVisibleChange || verification?.verdict === 'no-visible-change')
805
+ ? stateDelta(beforeScreen?.entry, afterReading?.entry)
806
+ : null;
807
+ if (changedState) verification = stateChanged(verification, changedState);
808
+ const wentNowhere = !changedState
809
+ && (Boolean(settled?.noVisibleChange) || verification?.verdict === 'no-visible-change');
792
810
  const aimNote = wentNowhere && aim.at
793
811
  ? screenmap.describePoint(
794
812
  // `beforeScreen` is null with verification off, and that is exactly
@@ -802,7 +820,9 @@ export async function runScript(
802
820
  const note = launchNote
803
821
  + (filling ? ` [settled, but ${filling} — waitFor content, do not act on this yet]` : '')
804
822
  + (settled?.smallChange ? ' [a small change, in one region only]' : '')
805
- + (settled?.noVisibleChange ? ' [no visible change]' : '')
823
+ + (changedState
824
+ ? ` [the screen did not move, but ${changedState.detail} — this worked, do not retry]`
825
+ : (settled?.noVisibleChange ? ' [no visible change]' : ''))
806
826
  // After the symptom, because it is the explanation of it.
807
827
  + (aimNote ? ` [${aimNote}]` : '')
808
828
  + (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
@@ -1594,6 +1614,37 @@ async function confirmNoChange(deviceQuery, verification, { beforeScreen, option
1594
1614
  };
1595
1615
  }
1596
1616
 
1617
+ /**
1618
+ * A control that changed state is not a step that did nothing.
1619
+ *
1620
+ * The sibling of `belowThreshold` on the same verdict, for the other kind of
1621
+ * evidence. `no-visible-change` arrives by two routes — the pixel detector and
1622
+ * the fingerprint agreeing we are on the same screen — and a switch or a radio
1623
+ * satisfies both while having plainly worked. Measured on a device: a radio
1624
+ * moving from "Top" to "Inline" reported `(settled 2714ms)
1625
+ * [no-visible-change]` with `selected` moving between the two rows in the
1626
+ * element list either side of it.
1627
+ *
1628
+ * This is the verdict an agent reads to decide whether to retry, and after item
1629
+ * 148 made the tap actually land, a retry undoes the thing that worked. So the
1630
+ * verdict carries the evidence rather than contradicting it.
1631
+ *
1632
+ * Still not `ok`: the screen genuinely did not become a different screen, and
1633
+ * claiming a transition that did not happen would be the opposite error. It
1634
+ * becomes `unverified` — the verdict this project uses for "something happened
1635
+ * and nobody can vouch for what" — with the change named.
1636
+ */
1637
+ export function stateChanged(verification, delta) {
1638
+ if (verification?.verdict !== 'no-visible-change' || !delta) return verification;
1639
+ return {
1640
+ ...verification,
1641
+ verdict: 'unverified',
1642
+ detail: `the screen did not change, but ${delta.detail}`
1643
+ + ' — the action worked; do not retry it, or you will undo it',
1644
+ stateDelta: delta,
1645
+ };
1646
+ }
1647
+
1597
1648
  export function belowThreshold(verification, settled) {
1598
1649
  if (verification?.verdict !== 'no-visible-change' || !settled?.smallChange) return verification;
1599
1650
  return {
@@ -1641,6 +1692,37 @@ async function focusHint(deviceQuery, target, ctx) {
1641
1692
  }
1642
1693
  }
1643
1694
 
1695
+ /**
1696
+ * Is anything focused? Asked before typing blind.
1697
+ *
1698
+ * `{"type": "text"}` with no `into` types into whatever has focus, and reported
1699
+ * `ok … [unconfirmed — no field named, so nothing was read back]`. The note was
1700
+ * honest and it was attached to a **green step**, so a chained `sim_do`
1701
+ * continued on a premise that had never been checked — a field report lost text
1702
+ * that went nowhere exactly this way.
1703
+ *
1704
+ * If nothing has focus, the keystrokes are lost and there is no version of that
1705
+ * which is an `ok`. The tree already answers it: one accessibility read, no
1706
+ * OCR, which is the same cost `focusHint` pays on the `into` path.
1707
+ *
1708
+ * Advisory in one direction only. A tree that cannot answer — no targets, a
1709
+ * read that threw — returns `unknown`, and the step proceeds with its
1710
+ * unconfirmed note rather than failing on our inability to look. Refusing to
1711
+ * type because we could not read the screen would be a worse trade than the
1712
+ * bug.
1713
+ */
1714
+ async function focusedSomewhere(deviceQuery, ctx) {
1715
+ try {
1716
+ const { entry } = await api.readScreenWith(deviceQuery, { useOcr: false, options: ctx.options });
1717
+ const targets = entry?.targets ?? [];
1718
+ if (!targets.length) return { known: false };
1719
+ const hit = targets.find((t) => t.focused === true);
1720
+ return { known: true, focused: Boolean(hit), where: hit?.label ?? hit?.identifier ?? null };
1721
+ } catch {
1722
+ return { known: false };
1723
+ }
1724
+ }
1725
+
1644
1726
  /**
1645
1727
  * How long to wait for the tap to become focus before inserting text.
1646
1728
  *
@@ -2319,11 +2401,29 @@ async function runStep(deviceQuery, udid, step, ctx) {
2319
2401
  journalWrite(udid, step, sent, back, ctx);
2320
2402
  return `typed into ${field.where}${back.note}${back.landed ? '' : field.quiet}`;
2321
2403
  }
2404
+ // **Refusing here was tried and reverted — see DEFERRED 161.** The idea
2405
+ // was that "nothing focused" is knowable before typing, so typing into
2406
+ // nothing should fail rather than report `ok`. On a device it produced a
2407
+ // FALSE REFUSAL on the legitimate case: after tapping Settings' own
2408
+ // search field, every element in the tree read `focused: false`. iOS does
2409
+ // not reliably publish AXFocused here, which `focusHint` above already
2410
+ // says in its own comment — *"with a hardware keyboard attached to the
2411
+ // simulator no software keyboard appears, so there is often nothing to
2412
+ // see"*. A gate on an unpublished signal blocks real work, which is worse
2413
+ // than the verdict it was fixing.
2414
+ //
2415
+ // So the focus reading stays advisory and goes into the note. What the
2416
+ // field report actually objected to was an honest caveat attached to a
2417
+ // GREEN step, and the note is where that has to be fixed until there is a
2418
+ // signal worth gating on.
2419
+ const focus = await focusedSomewhere(deviceQuery, ctx);
2322
2420
  await input.typeText(udid, step.text ?? step.value);
2323
- // No selector, so there is nothing to read back — which is a fine trade
2324
- // for typing into whatever Safari's own form chevrons focused, but it
2325
- // must not be reported as though the text was seen to land.
2326
- return 'typed text [unconfirmed — no field named, so nothing was read back]';
2421
+ return 'typed text [NOT CONFIRMED: no field named, so nothing was read back'
2422
+ + (focus.known && !focus.focused
2423
+ ? ' and nothing on this screen reports keyboard focus — the text may have gone nowhere'
2424
+ : '')
2425
+ + '. Name the field — {"type": {"into": "<field>", "text": "..."}} — to have it tapped'
2426
+ + ' first and read back]';
2327
2427
  }
2328
2428
  case 'paste': {
2329
2429
  // Long strings are much faster on the pasteboard than through the
@@ -2850,6 +2950,66 @@ async function runStep(deviceQuery, udid, step, ctx) {
2850
2950
  }
2851
2951
  }
2852
2952
 
2953
+ /**
2954
+ * What changed in the controls, when the pixels say nothing did.
2955
+ *
2956
+ * `settle()` decides `noVisibleChange` from frame hashes alone, and a control
2957
+ * that flips is eight times below the change threshold (DEFERRED 4, measured on
2958
+ * a switch: 0.00049 peak against 0.004). So a switch that genuinely flipped is
2959
+ * reported as an action that did nothing — with the changed value printed in the
2960
+ * same response.
2961
+ *
2962
+ * **That verdict became dangerous the moment item 148 made the tap land.** While
2963
+ * the tap missed the frame centre, `no-visible-change` was true and harmless.
2964
+ * Now the tap actuates, and `no-visible-change` is what an agent reads to decide
2965
+ * whether to retry — so a retry silently un-flips the toggle. An external
2966
+ * tester made exactly that point about my own commit message, which had claimed
2967
+ * the verdict "was telling the truth".
2968
+ *
2969
+ * This is wiring, not calibration. The element values are already in hand on
2970
+ * both sides of the action — `beforeScreen.entry` and the after-reading are read
2971
+ * for verification anyway — so the comparison costs nothing on the device. A
2972
+ * control whose value or selected state changed did something, whatever the
2973
+ * frame hash says.
2974
+ *
2975
+ * Identity across the two readings goes by identifier, then label, then
2976
+ * type-and-place: a row that moved is not the same reading of the same control,
2977
+ * and pretending otherwise would invent changes.
2978
+ */
2979
+ export function stateDelta(beforeEntry, afterEntry) {
2980
+ const keyOf = (t) => t.identifier || t.label || `${t.type ?? '?'}@${Math.round(t.x ?? 0)},${Math.round(t.y ?? 0)}`;
2981
+ const stateOf = (t) => `${t.value ?? ''}|${t.selected ?? ''}`;
2982
+ const before = new Map();
2983
+ for (const t of beforeEntry?.targets ?? []) {
2984
+ if (t.value == null && t.selected == null) continue;
2985
+ before.set(keyOf(t), { state: stateOf(t), target: t });
2986
+ }
2987
+ if (!before.size) return null;
2988
+ const changes = [];
2989
+ for (const t of afterEntry?.targets ?? []) {
2990
+ if (t.value == null && t.selected == null) continue;
2991
+ const was = before.get(keyOf(t));
2992
+ if (!was || was.state === stateOf(t)) continue;
2993
+ changes.push({
2994
+ name: t.label || t.identifier || keyOf(t),
2995
+ from: was.target.value ?? was.target.selected,
2996
+ to: t.value ?? t.selected,
2997
+ });
2998
+ }
2999
+ if (!changes.length) return null;
3000
+ // Name the control that turned ON, not the one that turned off. A radio move
3001
+ // changes two rows and reporting "Inline changed from true to false" is
3002
+ // accurate and reads like a loss — the useful half is what is selected now.
3003
+ const on = (v) => v === true || v === 'true' || v === '1';
3004
+ const first = changes.find((c) => on(c.to) && !on(c.from)) ?? changes[0];
3005
+ return {
3006
+ count: changes.length,
3007
+ detail: `${JSON.stringify(String(first.name).slice(0, 32))} changed from `
3008
+ + `${JSON.stringify(String(first.from))} to ${JSON.stringify(String(first.to))}`
3009
+ + (changes.length > 1 ? ` (and ${changes.length - 1} other control(s))` : ''),
3010
+ };
3011
+ }
3012
+
2853
3013
  /**
2854
3014
  * Why a settle gave up, in the numbers that decided it.
2855
3015
  *