simframe 0.10.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +181 -2
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
  9. package/native/simframed/Sources/simframed/main.swift +33 -3
  10. package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
  11. package/native/supervise.swift +216 -0
  12. package/package.json +4 -1
  13. package/scripts/check-package.mjs +22 -2
  14. package/scripts/check-private.mjs +9 -0
  15. package/scripts/ci-integration-local.sh +79 -0
  16. package/scripts/ci-memory.mjs +104 -20
  17. package/scripts/collect-rulings.mjs +312 -0
  18. package/scripts/eval-fingerprint.mjs +100 -23
  19. package/scripts/eval-perception.mjs +33 -0
  20. package/scripts/phase17-corpus.mjs +176 -0
  21. package/scripts/probe-network.mjs +118 -0
  22. package/scripts/soak-capture.mjs +72 -0
  23. package/skills/simframe/SKILL.md +257 -7
  24. package/src/actions.js +1791 -44
  25. package/src/cli.js +216 -14
  26. package/src/control.js +1 -0
  27. package/src/fingerprint.js +43 -1
  28. package/src/graph.js +136 -7
  29. package/src/index.js +252 -14
  30. package/src/input.js +66 -3
  31. package/src/localhelper.js +161 -0
  32. package/src/matching.js +64 -1
  33. package/src/mcp.js +333 -32
  34. package/src/metrics.js +148 -3
  35. package/src/ocr.js +18 -1
  36. package/src/planner.js +195 -0
  37. package/src/platform/android.js +25 -1
  38. package/src/platform/index.js +11 -1
  39. package/src/platform/ios.js +25 -1
  40. package/src/png.js +26 -0
  41. package/src/refs.js +51 -8
  42. package/src/regions.js +215 -1
  43. package/src/screenmap.js +89 -9
  44. package/src/supervisor.js +161 -0
  45. package/src/view.js +375 -11
  46. package/src/vocabulary.js +134 -0
  47. package/src/wrote.js +136 -0
@@ -56,6 +56,36 @@ function check(ok, label, detail = '') {
56
56
  return ok;
57
57
  }
58
58
 
59
+ let skipped = 0;
60
+
61
+ /**
62
+ * A check whose *setup* did not happen, reported as untested rather than failed.
63
+ *
64
+ * This exists because the harness did to itself what three peer reports spent a
65
+ * day telling us not to do to callers. A `simctl launch` timed out on a loaded
66
+ * runner, `allowFail` swallowed it, and the next check announced `FAIL the
67
+ * screen actually changed before testing the stale ref — 7070f77757 ->
68
+ * 7070f77757`. Every word of that is true and it names the wrong thing: the
69
+ * screen did not change because **the app never launched**, which the run knew
70
+ * and did not say. Two more checks failed downstream of the same cause.
71
+ *
72
+ * A skip does not fail the build, and that is deliberate. A red build caused by
73
+ * somebody else's build farm is the cry-wolf failure this project keeps writing
74
+ * down: it trains everyone to re-run rather than to read. But it is counted and
75
+ * printed, because a run that tested less than it claims must say so.
76
+ */
77
+ function skip(label, why) {
78
+ skipped += 1;
79
+ console.log(`skip ${label} — NOT TESTED: ${why}`);
80
+ return false;
81
+ }
82
+
83
+ /** Did a setup flow actually do what it was there for? */
84
+ function ran(res) {
85
+ if (!res || res.ok === false) return false;
86
+ return !(res.steps ?? []).some((st) => st.ok === false);
87
+ }
88
+
59
89
  /**
60
90
  * `expectFail` asserts a non-zero exit; `allowFail` tolerates one.
61
91
  *
@@ -96,7 +126,21 @@ async function json(args, opts) {
96
126
  * rather than serving a stale frame, which is the right behaviour and a
97
127
  * transient condition. Retry those, and only those.
98
128
  */
129
+ // Capture dropping out, and — since 2026-09-11 — simctl timing out.
130
+ //
131
+ // The second one is this repo's oldest CI complaint and it was never in this
132
+ // pattern, so `jsonRetry` sailed past it: `simctl launch` takes 47-55s per
133
+ // attempt on a loaded hosted runner and `simctl openurl` times out internally,
134
+ // which means simframe is handed a failure it did not cause and cannot fix.
135
+ // Three checks in one run failed downstream of exactly that.
136
+ //
137
+ // Retried HERE and deliberately not inside simframe, which is the rule DEFERRED
138
+ // already wrote down for the bench script: a retried launch is an action that
139
+ // fires twice, and the verify barrier exists to stop simframe doing that on its
140
+ // own initiative. A test harness re-running its own setup is a different thing
141
+ // from a driver silently repeating a user's action.
99
142
  const TRANSIENT = /did not produce a frame|display surface could not be read|no frames buffered/i;
143
+ const SIMCTL_FLAKE = /simctl|Command failed: xcrun|timed out/i;
100
144
 
101
145
  async function jsonRetry(args, opts, attempts = 3) {
102
146
  let last;
@@ -105,9 +149,13 @@ async function jsonRetry(args, opts, attempts = 3) {
105
149
  return await json(args, opts);
106
150
  } catch (err) {
107
151
  last = err;
108
- if (!TRANSIENT.test(err.message)) throw err;
109
- console.log(` (capture dropped out; retrying \`${args.join(' ')}\`)`);
110
- await new Promise((r) => setTimeout(r, 1500));
152
+ const capture = TRANSIENT.test(err.message);
153
+ const simctl = SIMCTL_FLAKE.test(err.message);
154
+ if (!capture && !simctl) throw err;
155
+ console.log(` (${capture ? 'capture dropped out' : 'simctl did not answer'}; retrying \`${args.join(' ')}\`)`);
156
+ // simctl's own timeouts are tens of seconds, so a 1.5s pause is not a
157
+ // wait, it is a formality. Give the runner room when that is the cause.
158
+ await new Promise((r) => setTimeout(r, simctl ? 5000 : 1500));
111
159
  }
112
160
  }
113
161
  // Out of attempts on a capture error: the device is not blinking, it is gone.
@@ -242,13 +290,15 @@ if (first) {
242
290
  // Now leave that screen WITHOUT re-reading it: `--json` skips the end-state
243
291
  // map, so the ref table still describes the screen we have left.
244
292
  const before = await markHash();
245
- await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
293
+ const left = await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
246
294
  const after = await markHash();
247
295
  // Two different apps are two different screens by construction. The pixel
248
296
  // hashes only have to agree with that, and they only get a say when they are
249
297
  // informative enough to have one.
250
298
  const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
251
- check(moved, 'the screen actually changed before testing the stale ref',
299
+ if (!ran(left)) skip('the screen actually changed before testing the stale ref',
300
+ 'the second app never launched, so there was no screen change to test against');
301
+ else check(moved, 'the screen actually changed before testing the stale ref',
252
302
  `${before.slice(0, 10)} -> ${after.slice(0, 10)}`
253
303
  + (informativeHash(before) && informativeHash(after) ? '' : ' (degenerate hash: not evidence either way)'));
254
304
  // Only assert the guard if the precondition actually held. Running it anyway
@@ -256,17 +306,28 @@ if (first) {
256
306
  // screen, which is a false accusation against the one layer this file exists
257
307
  // to defend — and it is how this check has failed twice.
258
308
  if (moved) {
259
- const stale = await cli(['find', `#${first.ref}`], { expectFail: true });
260
- // Matched on prose, which is this check's weakness: simframe refused
261
- // correctly with "#1 cannot be trusted here — simframe does not recognise
262
- // this screen. Read it again", and the check failed because that wording was
263
- // not one of the two it knew. The refusal is what matters, so the
264
- // alternation covers how a refusal is actually phrased; the durable fix is a
265
- // machine-readable reason on the failure, which `find --json` does not yet
266
- // carry.
267
- check(/different screen|read the screen|read it again|does not recognise this screen|cannot be trusted/i.test(stale),
309
+ // This matched on prose twice and went red twice, both times for a refusal
310
+ // that was correct and better worded than the alternation knew — most
311
+ // recently `"Welcome to Reminders" is not on this screen`, which refuses
312
+ // *and* names what the number stood for. `find --json` now carries the
313
+ // reason as a field, so the check reads the contract instead of the
314
+ // sentence. What is under test is unchanged: the ref must not resolve to
315
+ // the coordinates it was numbered at on the screen we have left.
316
+ // A refusal that cannot be parsed is a failed check, not a dead script:
317
+ // `json` throws on anything non-JSON reaching the stream, and this is the
318
+ // one call site that expects a failure, so it is the one that would take
319
+ // the whole file down with it.
320
+ let stale;
321
+ try {
322
+ stale = await json(['find', `#${first.ref}`], { expectFail: true });
323
+ } catch (err) {
324
+ stale = { ok: null, error: err.message };
325
+ }
326
+ const refused = stale.ok === false
327
+ && (stale.staleRef === true || stale.reason === 'unknown_screen' || stale.reason === 'ambiguous_intent');
328
+ check(refused,
268
329
  'a ref numbered on another screen refuses instead of tapping those coordinates',
269
- stale.trim().split('\n')[0]?.slice(0, 90));
330
+ `${stale.reason ?? 'no reason'}${stale.staleRef ? ' staleRef' : ''} — ${String(stale.error ?? '').split('\n')[0].slice(0, 70)}`);
270
331
  }
271
332
  } else {
272
333
  check(false, 'element refs', 'no elements to number');
@@ -343,7 +404,15 @@ const novelVerdicts = novelSteps.length
343
404
  // accusing the layer underneath it. A device that crashed SpringBoard mid-run
344
405
  // has not told us anything about the graph.
345
406
  const novelRan = novelSteps.length > 0 && novelSteps.every((r) => r.ok !== false);
346
- check(novelRan, 'the novel action ran at all', `[${novelVerdicts.join(', ')}]`);
407
+ // The comment above states the principle and this line used to contradict it:
408
+ // it called `check`, so a `simctl openurl` that timed out on the runner failed
409
+ // the build and said "the novel action ran at all" as though the graph were at
410
+ // fault. It is a precondition. Untested is not broken.
411
+ if (!novelRan) {
412
+ skip('the novel action ran at all', `the action could not be dispatched — [${novelVerdicts.join(', ')}]`);
413
+ } else {
414
+ check(true, 'the novel action ran at all', `[${novelVerdicts.join(', ')}]`);
415
+ }
347
416
  // The other half of the precondition, which was written above as a comment and
348
417
  // then trusted. It is not trustworthy: the positioning run sends the device
349
418
  // home, and a simulator that has been driven hard stops delivering `home` while
@@ -361,9 +430,20 @@ if (novelRan && novelMoved) {
361
430
  'an action never taken here before is reported as unverified, not as verified',
362
431
  `[${novelVerdicts.join(', ')}]`);
363
432
  }
364
- check(passes.some((p) => p.verdicts.includes('ok')),
365
- 'and once the graph has seen it, the outcome is predicted',
366
- `pass ${passes.findIndex((p) => p.verdicts.includes('ok')) + 1}`);
433
+ // The same precondition rule as the two above. The graph can only predict an
434
+ // outcome it has seen, and it can only have seen one if a pass actually ran —
435
+ // so a run in which every pass failed to dispatch says nothing about
436
+ // prediction. It failed the build as `pass 0` while the real cause was a
437
+ // simctl launch timing out, three checks upstream.
438
+ const anyPassRan = passes.some((p) => Array.isArray(p.run?.results) && p.run.results.some((r) => r.ok !== false));
439
+ if (!anyPassRan) {
440
+ skip('and once the graph has seen it, the outcome is predicted',
441
+ 'no pass dispatched a step, so the graph was never given anything to learn');
442
+ } else {
443
+ check(passes.some((p) => p.verdicts.includes('ok')),
444
+ 'and once the graph has seen it, the outcome is predicted',
445
+ `pass ${passes.findIndex((p) => p.verdicts.includes('ok')) + 1}`);
446
+ }
367
447
 
368
448
  // One direction only: a wrong turn must fail the run. The converse does not
369
449
  // hold — a run can fail for reasons that are not wrong turns, such as a step
@@ -472,5 +552,9 @@ try {
472
552
  }
473
553
  } catch { /* nothing to clean up */ }
474
554
 
475
- console.log(`\n${failures ? `${failures} check(s) failed` : 'every check passed'}`);
555
+ const summary = [
556
+ failures ? `${failures} check(s) failed` : 'every check passed',
557
+ skipped ? `${skipped} check(s) NOT TESTED — the runner could not set them up` : null,
558
+ ].filter(Boolean).join('; ');
559
+ console.log(`\n${summary}`);
476
560
  process.exit(failures ? 1 : 0);
@@ -0,0 +1,312 @@
1
+ #!/usr/bin/env node
2
+ // Generate a population of supervisor rulings, and score them.
3
+ //
4
+ // Items 101, 96, 106 and 109a all need rulings to replay, and the log held one,
5
+ // because a ruling requires a step that genuinely fails and Apple's own apps do
6
+ // not fail on command. The React Native testbed does, from a seeded stream, so
7
+ // this turns "we need dozens of rulings" into a script.
8
+ //
9
+ // Every fixture here has a **known correct outcome**, which is what makes the
10
+ // population scoreable rather than merely large:
11
+ //
12
+ // arriving a list still loading. Re-running the step works, so the right
13
+ // answer is wait/retry and the right outcome is `recovered`.
14
+ // blocked a required field is empty, so the thing waited for can never
15
+ // appear. Waiting and retrying are both wrong; `stopped` is right.
16
+ // refused a submit that failed. Re-running the *wait* cannot help — only
17
+ // re-submitting could, and that is the planner's call, not the
18
+ // supervisor's — so `stopped` is right here too.
19
+ //
20
+ // Note `refused` and `blocked` want the same answer for different reasons. That
21
+ // is deliberate: a judge that says stop for the wrong reason is still right, and
22
+ // a population that only contains obvious cases measures nothing.
23
+ //
24
+ // SIMFRAME_SUPERVISOR=apple node scripts/collect-rulings.mjs --device=<udid> --seeds=8
25
+ import { execFile } from 'node:child_process';
26
+ import * as actions from '../src/actions.js';
27
+ import * as api from '../src/index.js';
28
+ import * as metrics from '../src/metrics.js';
29
+ import * as store from '../src/store.js';
30
+ import * as supervisor from '../src/supervisor.js';
31
+
32
+ const arg = (name, fallback) => {
33
+ const hit = process.argv.find((a) => a.startsWith(`--${name}=`));
34
+ return hit ? hit.slice(name.length + 3) : fallback;
35
+ };
36
+ const device = arg('device');
37
+ const seeds = Number(arg('seeds', 6));
38
+ const BUNDLE = 'com.example.simframetestbed';
39
+
40
+ if (!supervisor.requested({})) {
41
+ console.error('SIMFRAME_SUPERVISOR is not set, so nothing would be judged and no ruling would be recorded.');
42
+ process.exit(2);
43
+ }
44
+
45
+ // Before anything else, because the device may already be wedged from the last
46
+ // run — `ensureDaemon` throws on a display that has stopped rendering, and it
47
+ // threw here on the third attempt of the afternoon, before the loop's own check
48
+ // could ever run. Driving one simulator hard for a few minutes is what does it.
49
+ {
50
+ const { execFileSync } = await import('node:child_process');
51
+ try {
52
+ await api.ensureDaemon(device);
53
+ } catch {
54
+ console.log('the device is not producing frames; reviving before starting');
55
+ try {
56
+ execFileSync(process.execPath, ['src/cli.js', 'revive', `--device=${device}`],
57
+ { cwd: new URL('..', import.meta.url).pathname, timeout: 300_000, stdio: 'inherit' });
58
+ } catch { /* reported below by the throw from ensureDaemon */ }
59
+ }
60
+ }
61
+
62
+ const { device: dev } = await api.ensureDaemon(device);
63
+ console.log(`device: ${dev.name} (${dev.runtime})`);
64
+
65
+ /**
66
+ * Cold-launch the app with a seed.
67
+ *
68
+ * `openUrl` on a terminated app launches it with that URL as its initial URL,
69
+ * which is the only way to set the seed *before* the first screen mounts and
70
+ * starts its own timers. Relaunching and then opening the URL would be too
71
+ * late: the list's delay has already been drawn from the default stream.
72
+ */
73
+ /**
74
+ * Scaffolding runs with the supervisor OFF, and that is not a detail.
75
+ *
76
+ * The first collection run produced 18 rulings of which **14 came from the
77
+ * harness's own plumbing** — seven from tapping an "Open in …?" dialog that was
78
+ * not always there, two from terminating an app that was not running, three
79
+ * from typing into a form we had failed to reach. Every one was a real
80
+ * consultation and every one would have landed in the population that 101 and
81
+ * 96 are going to measure.
82
+ *
83
+ * A fixture is a claim about what the supervisor should say. Plumbing is not,
84
+ * and a harness that cannot tell them apart is measuring itself.
85
+ */
86
+ const SCAFFOLD = { supervisor: 'none' };
87
+
88
+ const launchSeeded = async (seed) => {
89
+ // Terminating an app that is not running fails, and a failed step aborts the
90
+ // rest of the batch — so it gets its own call and its own shrug.
91
+ await actions.runScript(device, { steps: [{ terminate: BUNDLE }], verify: false, options: SCAFFOLD }).catch(() => null);
92
+ await actions.runScript(device, {
93
+ steps: [{ openUrl: `simframetestbed://seed/${seed}` }, { pause: 1200 }],
94
+ verify: false,
95
+ options: SCAFFOLD,
96
+ });
97
+ // iOS asks "Open in ...?" whenever a custom scheme is opened by another
98
+ // process, springboard included, and it asks on every launch. Answered here
99
+ // rather than designed around: the alternative is launch arguments, which RN
100
+ // does not expose to JavaScript without a native module.
101
+ await actions.runScript(device, {
102
+ steps: [{ tap: 'Open' }, { pause: 2200 }],
103
+ verify: false,
104
+ options: SCAFFOLD,
105
+ }).catch(async () => {
106
+ // No dialog this time. Give the app the same settling time anyway, so the
107
+ // fixture's timing does not depend on whether iOS felt like asking.
108
+ await actions.runScript(device, { steps: [{ pause: 2200 }], verify: false, options: SCAFFOLD }).catch(() => null);
109
+ });
110
+ };
111
+
112
+ /**
113
+ * Walk to where a fixture starts, unjudged — and *prove* you arrived.
114
+ *
115
+ * Navigation is scaffolding too. Three of the first run's stray rulings were a
116
+ * `type` that failed because the form had never been reached.
117
+ *
118
+ * The `arrive` half is the harder lesson, and it cost a wrong number. A walk
119
+ * that does not throw is not a walk that arrived: two `Next` taps landed on a
120
+ * live button, threw nothing, and advanced nothing, so **three of four rulings
121
+ * in the first clean run were taken on step 1 of a three-step form** while the
122
+ * fixture claimed they were about the review step. The scoreboard read 25% and
123
+ * was measuring the harness.
124
+ *
125
+ * `eval-fingerprint.mjs` already learned exactly this — it checks that each
126
+ * reading was taken on the screen the tour named, having once measured a
127
+ * distribution against readings taken somewhere else. The check simply had not
128
+ * been carried over.
129
+ */
130
+ const walk = async (steps, arrive) => {
131
+ try {
132
+ if (steps.length) await actions.runScript(device, { steps, verify: false, options: SCAFFOLD });
133
+ if (!arrive) return true;
134
+ // Asserted, not assumed. An assert that throws means we are not there.
135
+ await actions.runScript(device, {
136
+ steps: [{ assert: { value: arrive, is: 'visible' } }],
137
+ verify: false,
138
+ options: SCAFFOLD,
139
+ });
140
+ return true;
141
+ } catch {
142
+ return false;
143
+ }
144
+ };
145
+
146
+ /**
147
+ * Revive the device if capture has given up, and keep going.
148
+ *
149
+ * Collecting a population means driving one simulator hard for several minutes,
150
+ * and the display stops rendering when you do — twice in one afternoon here. The
151
+ * daemon detects it, tries both its recoveries, reports `stalled` and stops,
152
+ * because a capture loop that rebooted the device it was watching would be a
153
+ * tool reaching for the mains. This is a harness, not the product, and the
154
+ * operator's answer is exactly what it is here to automate — otherwise a run of
155
+ * twenty fixtures ends at the third and the population is however many rulings
156
+ * the device survived.
157
+ */
158
+ const reviveIfWedged = async () => {
159
+ const health = store.captureHealth(dev.udid);
160
+ if (!health?.stalled) return false;
161
+ process.stdout.write(' (capture stalled — reviving the device before continuing)\n');
162
+ await new Promise((resolve) => {
163
+ execFile(process.execPath, ['src/cli.js', 'revive', `--device=${dev.udid}`],
164
+ { timeout: 300_000, cwd: new URL('..', import.meta.url).pathname }, () => resolve());
165
+ });
166
+ return true;
167
+ };
168
+
169
+ const FIXTURES = [
170
+ {
171
+ name: 'arriving',
172
+ want: 'recovered',
173
+ // Pull to refresh rather than relying on the launch, because reaching the
174
+ // fixture now takes three seconds of its own — answering iOS's "Open in …?"
175
+ // — and by then the list has always arrived. The first clean run produced
176
+ // *no rulings at all* from this fixture for that reason. A refresh empties
177
+ // the list and reloads it on a fresh seeded delay, right where we want it.
178
+ walk: [{ swipe: { from: [201, 300], to: [201, 620] } }],
179
+ arrive: null,
180
+ judge: { waitFor: { value: 'Monstera #1', timeoutMs: 700 } },
181
+ expect: 'the list is still loading; its rows arrive shortly after launch',
182
+ },
183
+ {
184
+ name: 'detail',
185
+ want: 'recovered',
186
+ // A detail screen that is still fetching. Waiting is the right answer and
187
+ // re-running the step proves it, which is what makes this scoreable.
188
+ walk: [{ tap: 'Monstera #1' }],
189
+ arrive: null,
190
+ judge: { waitFor: { value: 'Prefers bright indirect light', timeoutMs: 700 } },
191
+ expect: 'the detail screen is still fetching; its text arrives shortly',
192
+ },
193
+ {
194
+ name: 'secondwave',
195
+ want: 'recovered',
196
+ // The list renders its count header, then a third of its rows, then the
197
+ // rest. `Jade #24` is in the last wave, so a tight wait for it fails while
198
+ // the screen is *stable and incomplete at the same moment* — the state that
199
+ // has fooled settle detection and the supervisor alike.
200
+ walk: [{ swipe: { from: [201, 300], to: [201, 620] } }],
201
+ arrive: null,
202
+ judge: { waitFor: { value: 'Jade #24', timeoutMs: 700 } },
203
+ expect: 'the list arrives in waves and this row is in the last one',
204
+ },
205
+ {
206
+ name: 'blocked',
207
+ want: 'stopped',
208
+ walk: [
209
+ { tap: 'Forms, tab, 2 of 3' }, { pause: 900 }, { tap: 'Stepped form' }, { pause: 900 },
210
+ { tap: 'Next' }, { pause: 1200 }, { tap: 'Next' }, { pause: 1200 },
211
+ ],
212
+ // The review step, proved rather than hoped for.
213
+ arrive: 'Step 3 of 3',
214
+ judge: { waitFor: { value: 'Submitted', timeoutMs: 2500 } },
215
+ expect: 'Review is blocked until Species is filled in, and it is empty',
216
+ },
217
+ {
218
+ name: 'refused',
219
+ want: 'stopped',
220
+ walk: [
221
+ { tap: 'Forms, tab, 2 of 3' }, { pause: 900 }, { tap: 'One-step form' },
222
+ { pause: 6500 },
223
+ { type: { into: 'Your Name', text: 'Ada' } },
224
+ { tap: 'Submit' }, { pause: 1200 },
225
+ ],
226
+ // The rejection is on screen, so the submit demonstrably happened and
227
+ // failed — otherwise this fixture can pass by never having submitted.
228
+ arrive: 'The order was rejected',
229
+ // "Saved" was the string here and it fuzzy-matched "Could not save: …", so
230
+ // the judge step *succeeded* on the failure it was meant to catch and the
231
+ // fixture produced no rulings at all. Success and failure now share no words.
232
+ judge: { waitFor: { value: 'Order placed', timeoutMs: 2500 } },
233
+ expect: 'the first submit always fails and the second works, so waiting cannot help',
234
+ },
235
+ ];
236
+
237
+ const before = metrics.readSupervisions(dev.udid).length;
238
+ let runs = 0;
239
+ let skipped = 0;
240
+
241
+ for (let i = 0; i < seeds; i += 1) {
242
+ const seed = 1000 + i * 7;
243
+ for (const fx of FIXTURES) {
244
+ await reviveIfWedged();
245
+ await launchSeeded(seed);
246
+ const reached = await walk(fx.walk, fx.arrive);
247
+ if (!reached) {
248
+ process.stdout.write(` seed ${seed} ${fx.name.padEnd(9)} SKIPPED — could not reach the fixture\n`);
249
+ skipped += 1;
250
+ continue;
251
+ }
252
+ try {
253
+ await actions.runScript(device, {
254
+ steps: [{ ...fx.judge, expect: fx.expect }],
255
+ supervise: fx.name,
256
+ verify: true,
257
+ });
258
+ } catch { /* a failing step is the point */ }
259
+ runs += 1;
260
+ process.stdout.write(` seed ${seed} ${fx.name.padEnd(9)} judged\n`);
261
+ }
262
+ }
263
+
264
+ const all = metrics.readSupervisions(dev.udid);
265
+ const fresh = all.slice(before);
266
+ console.log(`\n${runs} judged step(s), ${skipped} skipped, ${fresh.length} ruling(s)\n`);
267
+
268
+ const byFixture = new Map(FIXTURES.map((f) => [f.expect, f]));
269
+ const score = new Map(FIXTURES.map((f) => [f.name, { n: 0, right: 0, decisions: {}, outcomes: {} }]));
270
+ let unattributed = 0;
271
+ for (const r of fresh) {
272
+ const fx = byFixture.get(r.expect);
273
+ if (!fx) { unattributed += 1; continue; }
274
+ const s = score.get(fx.name);
275
+ s.n += 1;
276
+ s.decisions[r.decision] = (s.decisions[r.decision] ?? 0) + 1;
277
+ s.outcomes[r.outcome] = (s.outcomes[r.outcome] ?? 0) + 1;
278
+ if (r.outcome === fx.want) s.right += 1;
279
+ }
280
+
281
+ console.log(`${'fixture'.padEnd(10)} ${'n'.padStart(3)} ${'right'.padStart(6)} wanted decisions / outcomes`);
282
+ for (const fx of FIXTURES) {
283
+ const s = score.get(fx.name);
284
+ const pct = s.n ? `${Math.round((100 * s.right) / s.n)}%` : '—';
285
+ console.log(
286
+ `${fx.name.padEnd(10)} ${String(s.n).padStart(3)} ${pct.padStart(6)} ${fx.want.padEnd(10)} `
287
+ + `${JSON.stringify(s.decisions)} / ${JSON.stringify(s.outcomes)}`,
288
+ );
289
+ }
290
+ if (unattributed) console.log(`\n${unattributed} ruling(s) came from a step this script did not label.`);
291
+
292
+ // Said before the accuracy is read, not after. The first population was 12
293
+ // `stop` to 2 `wait`, so a majority-class guess scored 86% against the model's
294
+ // 64% — and every number computed from it was an artifact of that skew. A
295
+ // scoreboard that prints accuracy without printing its own balance invites
296
+ // exactly that mistake a second time.
297
+ {
298
+ const want = {};
299
+ for (const fx of FIXTURES) {
300
+ const s = score.get(fx.name);
301
+ want[fx.want] = (want[fx.want] ?? 0) + s.n;
302
+ }
303
+ const total = Object.values(want).reduce((a, b) => a + b, 0);
304
+ const biggest = Math.max(0, ...Object.values(want));
305
+ const baseline = total ? Math.round((100 * biggest) / total) : 0;
306
+ console.log(`\nbalance: ${JSON.stringify(want)} — guessing the commonest answer scores ${baseline}%.`);
307
+ if (baseline > 65) {
308
+ console.log(' SKEWED. Any accuracy above is mostly a fact about the fixture set, not the judge.');
309
+ console.log(' Add fixtures for the under-represented answer before comparing anything against anything.');
310
+ }
311
+ }
312
+ console.log(`\nthe log now holds ${all.length} ruling(s) — simframe supervisions --device=${dev.udid}`);
@@ -87,6 +87,43 @@ const readings = [];
87
87
  /** Navigations that did not land before the reading was taken. */
88
88
  const arrivalFailures = [];
89
89
 
90
+ /**
91
+ * The tokens that carry a name, as opposed to a shape.
92
+ *
93
+ * A fingerprint is deliberately geometry — role, region, size, position — and
94
+ * chrome labels are the only text that survives into it (`fingerprint.js`).
95
+ * That module's own comment states the consequence: "two list screens with
96
+ * identical structure differ by their title, and nothing else says so". So a
97
+ * reading with none of these has no identity to speak of, and two such
98
+ * readings of *different* screens can hash identically. Counted here because
99
+ * that is diagnosable and "the tour went somewhere unintended" is not.
100
+ */
101
+ const namedTokens = (tokens) => tokens.filter((t) => t.includes('"'));
102
+
103
+ /**
104
+ * Write the readings out now, rather than after the checks.
105
+ *
106
+ * The write used to sit past every `process.exit(1)`, so the only run that
107
+ * kept its evidence was the run with nothing to explain. A failing run exited
108
+ * before the file existed and `if: always()` on the upload step faithfully
109
+ * uploaded nothing — which is how one red integration job cost an evening of
110
+ * inferring from a summary line while `analyse-fingerprint.mjs`, which exists
111
+ * to classify exactly these divergences, had no file to read. Called as soon
112
+ * as the tour is done and again with the analysis, so every exit below this
113
+ * point still leaves the readings behind.
114
+ */
115
+ const save = (extra = {}) => {
116
+ if (!outFile) return;
117
+ fs.writeFileSync(outFile, JSON.stringify({
118
+ label, device: dev.name, runtime: dev.runtime, rounds, at: Date.now(),
119
+ threshold: graph.SIMILARITY_THRESHOLD,
120
+ // Tokens are kept. They were stripped here once, and the first time the
121
+ // margin narrowed the run could not be diagnosed from its own output.
122
+ readings,
123
+ ...extra,
124
+ }, null, 2));
125
+ };
126
+
90
127
  for (let round = 1; round <= rounds; round += 1) {
91
128
  for (const screen of tour) {
92
129
  if (screen.steps?.length) {
@@ -136,12 +173,25 @@ for (let round = 1; round <= rounds; round += 1) {
136
173
  // like fingerprint drift.
137
174
  sources: id.entry?.sources ?? [],
138
175
  });
176
+ // Sensors and named tokens are on every line, not only inside a failure.
177
+ // Both were invisible until a run failed, and both are what the failure
178
+ // turns out to be about: two readings of one screen taken by different
179
+ // sensors do not share a hash by design, and a reading carrying no chrome
180
+ // label cannot be told from any other screen of the same shape. A summary
181
+ // line that hid those sent an evening after hosted-runner speed.
139
182
  process.stdout.write(
140
- ` round ${round} ${screen.name.padEnd(14)} ${String(id.hash).slice(0, 10)} ${String((id.tokens ?? []).length).padStart(3)} tokens${id.settled ? '' : ' (never settled)'}\n`,
183
+ ` round ${round} ${screen.name.padEnd(22)} ${String(id.hash).slice(0, 10)} `
184
+ + `${String((id.tokens ?? []).length).padStart(3)} tokens `
185
+ + `${String(namedTokens(id.tokens ?? []).length).padStart(2)} named `
186
+ + `${((id.entry?.sources ?? []).join('+') || 'none').padEnd(9)}`
187
+ + `${id.settled ? '' : ' (never settled)'}\n`,
141
188
  );
142
189
  }
143
190
  }
144
191
 
192
+ save();
193
+ if (outFile) console.log(`\nwrote ${readings.length} readings to ${outFile}`);
194
+
145
195
  if (arrivalFailures.length) {
146
196
  console.error(`\nFAIL ${arrivalFailures.length} reading(s) were taken on the previous screen:`);
147
197
  for (const f of arrivalFailures) console.error(` ${f}`);
@@ -178,11 +228,17 @@ function findStrays(all) {
178
228
  if (siblings.length < 2) continue;
179
229
  const bestSelf = Math.max(...siblings.map((o) => fingerprint.similarity(r.tokens, o.tokens)));
180
230
  const others = all.filter((o) => o.name !== r.name);
181
- const bestOther = others.length
182
- ? Math.max(...others.map((o) => fingerprint.similarity(r.tokens, o.tokens)))
183
- : 0;
231
+ // Which screen it resembles, not merely how much. A stray that resembles
232
+ // one particular other screen at 1.00 is a different animal from one that
233
+ // resembles everything weakly, and the report could not tell them apart.
234
+ let match = null;
235
+ let bestOther = 0;
236
+ for (const o of others) {
237
+ const s = fingerprint.similarity(r.tokens, o.tokens);
238
+ if (s > bestOther) { bestOther = s; match = o; }
239
+ }
184
240
  if (bestSelf < graph.SIMILARITY_THRESHOLD && bestOther >= bestSelf) {
185
- strays.push({ reading: r, bestSelf, bestOther });
241
+ strays.push({ reading: r, bestSelf, bestOther, match });
186
242
  }
187
243
  }
188
244
  return strays;
@@ -190,14 +246,41 @@ function findStrays(all) {
190
246
 
191
247
  const strays = findStrays(readings);
192
248
  if (strays.length) {
193
- console.error(`\nFAIL ${strays.length} reading(s) were taken on a screen other than the one named:`);
194
- for (const { reading, bestSelf, bestOther } of strays) {
195
- console.error(` ${reading.name} r${reading.round}: resembles its own screen ${bestSelf.toFixed(2)}, `
196
- + `another screen ${bestOther.toFixed(2)} (${reading.count} tokens, sources ${reading.sources.join('+') || 'none'})`);
249
+ console.error(`\nFAIL ${strays.length} reading(s) do not resemble their own screen:`);
250
+ // Two causes wear the same symptom, and until now the report asserted the
251
+ // second one. A reading can be unlike its siblings because the tour went
252
+ // somewhere unintended — or because the fingerprint could not tell two
253
+ // screens apart, which is the harness's actual subject. They are separable
254
+ // from the data in hand: a collision is a reading that carries no chrome
255
+ // label while matching one particular other screen almost exactly, and a
256
+ // wrong turn is one whose tokens name a screen the tour did not ask for.
257
+ let collisions = 0;
258
+ for (const { reading, bestSelf, bestOther, match } of strays) {
259
+ const named = namedTokens(reading.tokens);
260
+ const matchNamed = match ? namedTokens(match.tokens) : [];
261
+ const collided = bestOther >= 0.99 && named.length === 0 && matchNamed.length === 0;
262
+ if (collided) collisions += 1;
263
+ console.error(` ${reading.name} r${reading.round}: own screen ${bestSelf.toFixed(2)}, `
264
+ + `${match ? `${match.name} r${match.round}` : 'another screen'} ${bestOther.toFixed(2)} `
265
+ + `(${reading.count} tokens, ${named.length} named, sources ${reading.sources.join('+') || 'none'})`);
266
+ if (collided) {
267
+ console.error(' ^ a COLLISION, not a wrong turn: neither reading carries a chrome');
268
+ console.error(' label, so both are structure with no name and the fingerprint has');
269
+ console.error(' nothing left to tell two list screens apart.');
270
+ }
271
+ if (named.length) console.error(` names: ${named.map((t) => t.slice(t.indexOf('"'), t.lastIndexOf('"') + 1)).join(' ')}`);
272
+ }
273
+ if (collisions) {
274
+ console.error(`\n${collisions} of ${strays.length} are fingerprint collisions. That is this harness's own subject,`);
275
+ console.error('not a tour fault: a reading whose chrome label went missing cannot establish');
276
+ console.error('identity, and comparing it as though it could is what produced the verdict above.');
277
+ } else {
278
+ console.error('\nThat is the tour going somewhere unintended, not the fingerprint drifting, and');
279
+ console.error('measuring it as either distribution poisons both ends. Fix the tour — a tap that');
280
+ console.error('missed, or a screen that needs longer than its pause — and re-run.');
197
281
  }
198
- console.error('\nThat is the tour going somewhere unintended, not the fingerprint drifting, and');
199
- console.error('measuring it as either distribution poisons both ends. Fix the tour — a tap that');
200
- console.error('missed, or a screen that needs longer than its pause — and re-run.');
282
+ console.error(`\nEvery reading is in ${outFile ?? 'the --out file'}; `
283
+ + 'run `node scripts/analyse-fingerprint.mjs <that file>` to classify the divergent tokens.');
201
284
  process.exit(1);
202
285
  }
203
286
 
@@ -317,17 +400,11 @@ for (const r of readings) {
317
400
  }
318
401
  console.log(`\n${labels.size} distinct chrome label(s) entered identity: ${[...labels].sort().join(' · ') || '(none)'}`);
319
402
 
320
- if (outFile) {
321
- fs.writeFileSync(outFile, JSON.stringify({
322
- label, device: dev.name, runtime: dev.runtime, rounds, at: Date.now(),
323
- threshold, same: s, mixed: m, different: d, gap, separated, thresholdInGap,
324
- labels: [...labels].sort(),
325
- // Tokens are kept. They were stripped here, and the first time the margin
326
- // narrowed the run could not be diagnosed from its own output.
327
- readings,
328
- }, null, 2));
329
- console.log(`\nwrote ${outFile}`);
330
- }
403
+ save({
404
+ same: s, mixed: m, different: d, gap, separated, thresholdInGap,
405
+ labels: [...labels].sort(),
406
+ });
407
+ if (outFile) console.log(`\nwrote ${outFile}`);
331
408
 
332
409
  // The stated margin, checked rather than eyeballed. A person noticing that a
333
410
  // number moved is not a test; this is the machine that re-measures it.