simframe 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +181 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +33 -3
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +216 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/scripts/probe-network.mjs +118 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +257 -7
- package/src/actions.js +1791 -44
- package/src/cli.js +216 -14
- package/src/control.js +1 -0
- package/src/fingerprint.js +43 -1
- package/src/graph.js +136 -7
- package/src/index.js +252 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +161 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +333 -32
- package/src/metrics.js +148 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +25 -1
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +25 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +215 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +161 -0
- package/src/view.js +375 -11
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/scripts/ci-memory.mjs
CHANGED
|
@@ -56,6 +56,36 @@ function check(ok, label, detail = '') {
|
|
|
56
56
|
return ok;
|
|
57
57
|
}
|
|
58
58
|
|
|
59
|
+
let skipped = 0;
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* A check whose *setup* did not happen, reported as untested rather than failed.
|
|
63
|
+
*
|
|
64
|
+
* This exists because the harness did to itself what three peer reports spent a
|
|
65
|
+
* day telling us not to do to callers. A `simctl launch` timed out on a loaded
|
|
66
|
+
* runner, `allowFail` swallowed it, and the next check announced `FAIL the
|
|
67
|
+
* screen actually changed before testing the stale ref — 7070f77757 ->
|
|
68
|
+
* 7070f77757`. Every word of that is true and it names the wrong thing: the
|
|
69
|
+
* screen did not change because **the app never launched**, which the run knew
|
|
70
|
+
* and did not say. Two more checks failed downstream of the same cause.
|
|
71
|
+
*
|
|
72
|
+
* A skip does not fail the build, and that is deliberate. A red build caused by
|
|
73
|
+
* somebody else's build farm is the cry-wolf failure this project keeps writing
|
|
74
|
+
* down: it trains everyone to re-run rather than to read. But it is counted and
|
|
75
|
+
* printed, because a run that tested less than it claims must say so.
|
|
76
|
+
*/
|
|
77
|
+
function skip(label, why) {
|
|
78
|
+
skipped += 1;
|
|
79
|
+
console.log(`skip ${label} — NOT TESTED: ${why}`);
|
|
80
|
+
return false;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** Did a setup flow actually do what it was there for? */
|
|
84
|
+
function ran(res) {
|
|
85
|
+
if (!res || res.ok === false) return false;
|
|
86
|
+
return !(res.steps ?? []).some((st) => st.ok === false);
|
|
87
|
+
}
|
|
88
|
+
|
|
59
89
|
/**
|
|
60
90
|
* `expectFail` asserts a non-zero exit; `allowFail` tolerates one.
|
|
61
91
|
*
|
|
@@ -96,7 +126,21 @@ async function json(args, opts) {
|
|
|
96
126
|
* rather than serving a stale frame, which is the right behaviour and a
|
|
97
127
|
* transient condition. Retry those, and only those.
|
|
98
128
|
*/
|
|
129
|
+
// Capture dropping out, and — since 2026-09-11 — simctl timing out.
|
|
130
|
+
//
|
|
131
|
+
// The second one is this repo's oldest CI complaint and it was never in this
|
|
132
|
+
// pattern, so `jsonRetry` sailed past it: `simctl launch` takes 47-55s per
|
|
133
|
+
// attempt on a loaded hosted runner and `simctl openurl` times out internally,
|
|
134
|
+
// which means simframe is handed a failure it did not cause and cannot fix.
|
|
135
|
+
// Three checks in one run failed downstream of exactly that.
|
|
136
|
+
//
|
|
137
|
+
// Retried HERE and deliberately not inside simframe, which is the rule DEFERRED
|
|
138
|
+
// already wrote down for the bench script: a retried launch is an action that
|
|
139
|
+
// fires twice, and the verify barrier exists to stop simframe doing that on its
|
|
140
|
+
// own initiative. A test harness re-running its own setup is a different thing
|
|
141
|
+
// from a driver silently repeating a user's action.
|
|
99
142
|
const TRANSIENT = /did not produce a frame|display surface could not be read|no frames buffered/i;
|
|
143
|
+
const SIMCTL_FLAKE = /simctl|Command failed: xcrun|timed out/i;
|
|
100
144
|
|
|
101
145
|
async function jsonRetry(args, opts, attempts = 3) {
|
|
102
146
|
let last;
|
|
@@ -105,9 +149,13 @@ async function jsonRetry(args, opts, attempts = 3) {
|
|
|
105
149
|
return await json(args, opts);
|
|
106
150
|
} catch (err) {
|
|
107
151
|
last = err;
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
152
|
+
const capture = TRANSIENT.test(err.message);
|
|
153
|
+
const simctl = SIMCTL_FLAKE.test(err.message);
|
|
154
|
+
if (!capture && !simctl) throw err;
|
|
155
|
+
console.log(` (${capture ? 'capture dropped out' : 'simctl did not answer'}; retrying \`${args.join(' ')}\`)`);
|
|
156
|
+
// simctl's own timeouts are tens of seconds, so a 1.5s pause is not a
|
|
157
|
+
// wait, it is a formality. Give the runner room when that is the cause.
|
|
158
|
+
await new Promise((r) => setTimeout(r, simctl ? 5000 : 1500));
|
|
111
159
|
}
|
|
112
160
|
}
|
|
113
161
|
// Out of attempts on a capture error: the device is not blinking, it is gone.
|
|
@@ -242,13 +290,15 @@ if (first) {
|
|
|
242
290
|
// Now leave that screen WITHOUT re-reading it: `--json` skips the end-state
|
|
243
291
|
// map, so the ref table still describes the screen we have left.
|
|
244
292
|
const before = await markHash();
|
|
245
|
-
await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
|
|
293
|
+
const left = await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
|
|
246
294
|
const after = await markHash();
|
|
247
295
|
// Two different apps are two different screens by construction. The pixel
|
|
248
296
|
// hashes only have to agree with that, and they only get a say when they are
|
|
249
297
|
// informative enough to have one.
|
|
250
298
|
const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
|
|
251
|
-
|
|
299
|
+
if (!ran(left)) skip('the screen actually changed before testing the stale ref',
|
|
300
|
+
'the second app never launched, so there was no screen change to test against');
|
|
301
|
+
else check(moved, 'the screen actually changed before testing the stale ref',
|
|
252
302
|
`${before.slice(0, 10)} -> ${after.slice(0, 10)}`
|
|
253
303
|
+ (informativeHash(before) && informativeHash(after) ? '' : ' (degenerate hash: not evidence either way)'));
|
|
254
304
|
// Only assert the guard if the precondition actually held. Running it anyway
|
|
@@ -256,17 +306,28 @@ if (first) {
|
|
|
256
306
|
// screen, which is a false accusation against the one layer this file exists
|
|
257
307
|
// to defend — and it is how this check has failed twice.
|
|
258
308
|
if (moved) {
|
|
259
|
-
|
|
260
|
-
//
|
|
261
|
-
//
|
|
262
|
-
//
|
|
263
|
-
//
|
|
264
|
-
//
|
|
265
|
-
//
|
|
266
|
-
//
|
|
267
|
-
|
|
309
|
+
// This matched on prose twice and went red twice, both times for a refusal
|
|
310
|
+
// that was correct and better worded than the alternation knew — most
|
|
311
|
+
// recently `"Welcome to Reminders" is not on this screen`, which refuses
|
|
312
|
+
// *and* names what the number stood for. `find --json` now carries the
|
|
313
|
+
// reason as a field, so the check reads the contract instead of the
|
|
314
|
+
// sentence. What is under test is unchanged: the ref must not resolve to
|
|
315
|
+
// the coordinates it was numbered at on the screen we have left.
|
|
316
|
+
// A refusal that cannot be parsed is a failed check, not a dead script:
|
|
317
|
+
// `json` throws on anything non-JSON reaching the stream, and this is the
|
|
318
|
+
// one call site that expects a failure, so it is the one that would take
|
|
319
|
+
// the whole file down with it.
|
|
320
|
+
let stale;
|
|
321
|
+
try {
|
|
322
|
+
stale = await json(['find', `#${first.ref}`], { expectFail: true });
|
|
323
|
+
} catch (err) {
|
|
324
|
+
stale = { ok: null, error: err.message };
|
|
325
|
+
}
|
|
326
|
+
const refused = stale.ok === false
|
|
327
|
+
&& (stale.staleRef === true || stale.reason === 'unknown_screen' || stale.reason === 'ambiguous_intent');
|
|
328
|
+
check(refused,
|
|
268
329
|
'a ref numbered on another screen refuses instead of tapping those coordinates',
|
|
269
|
-
stale.
|
|
330
|
+
`${stale.reason ?? 'no reason'}${stale.staleRef ? ' staleRef' : ''} — ${String(stale.error ?? '').split('\n')[0].slice(0, 70)}`);
|
|
270
331
|
}
|
|
271
332
|
} else {
|
|
272
333
|
check(false, 'element refs', 'no elements to number');
|
|
@@ -343,7 +404,15 @@ const novelVerdicts = novelSteps.length
|
|
|
343
404
|
// accusing the layer underneath it. A device that crashed SpringBoard mid-run
|
|
344
405
|
// has not told us anything about the graph.
|
|
345
406
|
const novelRan = novelSteps.length > 0 && novelSteps.every((r) => r.ok !== false);
|
|
346
|
-
|
|
407
|
+
// The comment above states the principle and this line used to contradict it:
|
|
408
|
+
// it called `check`, so a `simctl openurl` that timed out on the runner failed
|
|
409
|
+
// the build and said "the novel action ran at all" as though the graph were at
|
|
410
|
+
// fault. It is a precondition. Untested is not broken.
|
|
411
|
+
if (!novelRan) {
|
|
412
|
+
skip('the novel action ran at all', `the action could not be dispatched — [${novelVerdicts.join(', ')}]`);
|
|
413
|
+
} else {
|
|
414
|
+
check(true, 'the novel action ran at all', `[${novelVerdicts.join(', ')}]`);
|
|
415
|
+
}
|
|
347
416
|
// The other half of the precondition, which was written above as a comment and
|
|
348
417
|
// then trusted. It is not trustworthy: the positioning run sends the device
|
|
349
418
|
// home, and a simulator that has been driven hard stops delivering `home` while
|
|
@@ -361,9 +430,20 @@ if (novelRan && novelMoved) {
|
|
|
361
430
|
'an action never taken here before is reported as unverified, not as verified',
|
|
362
431
|
`[${novelVerdicts.join(', ')}]`);
|
|
363
432
|
}
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
433
|
+
// The same precondition rule as the two above. The graph can only predict an
|
|
434
|
+
// outcome it has seen, and it can only have seen one if a pass actually ran —
|
|
435
|
+
// so a run in which every pass failed to dispatch says nothing about
|
|
436
|
+
// prediction. It failed the build as `pass 0` while the real cause was a
|
|
437
|
+
// simctl launch timing out, three checks upstream.
|
|
438
|
+
const anyPassRan = passes.some((p) => Array.isArray(p.run?.results) && p.run.results.some((r) => r.ok !== false));
|
|
439
|
+
if (!anyPassRan) {
|
|
440
|
+
skip('and once the graph has seen it, the outcome is predicted',
|
|
441
|
+
'no pass dispatched a step, so the graph was never given anything to learn');
|
|
442
|
+
} else {
|
|
443
|
+
check(passes.some((p) => p.verdicts.includes('ok')),
|
|
444
|
+
'and once the graph has seen it, the outcome is predicted',
|
|
445
|
+
`pass ${passes.findIndex((p) => p.verdicts.includes('ok')) + 1}`);
|
|
446
|
+
}
|
|
367
447
|
|
|
368
448
|
// One direction only: a wrong turn must fail the run. The converse does not
|
|
369
449
|
// hold — a run can fail for reasons that are not wrong turns, such as a step
|
|
@@ -472,5 +552,9 @@ try {
|
|
|
472
552
|
}
|
|
473
553
|
} catch { /* nothing to clean up */ }
|
|
474
554
|
|
|
475
|
-
|
|
555
|
+
const summary = [
|
|
556
|
+
failures ? `${failures} check(s) failed` : 'every check passed',
|
|
557
|
+
skipped ? `${skipped} check(s) NOT TESTED — the runner could not set them up` : null,
|
|
558
|
+
].filter(Boolean).join('; ');
|
|
559
|
+
console.log(`\n${summary}`);
|
|
476
560
|
process.exit(failures ? 1 : 0);
|
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Generate a population of supervisor rulings, and score them.
|
|
3
|
+
//
|
|
4
|
+
// Items 101, 96, 106 and 109a all need rulings to replay, and the log held one,
|
|
5
|
+
// because a ruling requires a step that genuinely fails and Apple's own apps do
|
|
6
|
+
// not fail on command. The React Native testbed does, from a seeded stream, so
|
|
7
|
+
// this turns "we need dozens of rulings" into a script.
|
|
8
|
+
//
|
|
9
|
+
// Every fixture here has a **known correct outcome**, which is what makes the
|
|
10
|
+
// population scoreable rather than merely large:
|
|
11
|
+
//
|
|
12
|
+
// arriving a list still loading. Re-running the step works, so the right
|
|
13
|
+
// answer is wait/retry and the right outcome is `recovered`.
|
|
14
|
+
// blocked a required field is empty, so the thing waited for can never
|
|
15
|
+
// appear. Waiting and retrying are both wrong; `stopped` is right.
|
|
16
|
+
// refused a submit that failed. Re-running the *wait* cannot help — only
|
|
17
|
+
// re-submitting could, and that is the planner's call, not the
|
|
18
|
+
// supervisor's — so `stopped` is right here too.
|
|
19
|
+
//
|
|
20
|
+
// Note `refused` and `blocked` want the same answer for different reasons. That
|
|
21
|
+
// is deliberate: a judge that says stop for the wrong reason is still right, and
|
|
22
|
+
// a population that only contains obvious cases measures nothing.
|
|
23
|
+
//
|
|
24
|
+
// SIMFRAME_SUPERVISOR=apple node scripts/collect-rulings.mjs --device=<udid> --seeds=8
|
|
25
|
+
import { execFile } from 'node:child_process';
|
|
26
|
+
import * as actions from '../src/actions.js';
|
|
27
|
+
import * as api from '../src/index.js';
|
|
28
|
+
import * as metrics from '../src/metrics.js';
|
|
29
|
+
import * as store from '../src/store.js';
|
|
30
|
+
import * as supervisor from '../src/supervisor.js';
|
|
31
|
+
|
|
32
|
+
const arg = (name, fallback) => {
|
|
33
|
+
const hit = process.argv.find((a) => a.startsWith(`--${name}=`));
|
|
34
|
+
return hit ? hit.slice(name.length + 3) : fallback;
|
|
35
|
+
};
|
|
36
|
+
const device = arg('device');
|
|
37
|
+
const seeds = Number(arg('seeds', 6));
|
|
38
|
+
const BUNDLE = 'com.example.simframetestbed';
|
|
39
|
+
|
|
40
|
+
if (!supervisor.requested({})) {
|
|
41
|
+
console.error('SIMFRAME_SUPERVISOR is not set, so nothing would be judged and no ruling would be recorded.');
|
|
42
|
+
process.exit(2);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// Before anything else, because the device may already be wedged from the last
|
|
46
|
+
// run — `ensureDaemon` throws on a display that has stopped rendering, and it
|
|
47
|
+
// threw here on the third attempt of the afternoon, before the loop's own check
|
|
48
|
+
// could ever run. Driving one simulator hard for a few minutes is what does it.
|
|
49
|
+
{
|
|
50
|
+
const { execFileSync } = await import('node:child_process');
|
|
51
|
+
try {
|
|
52
|
+
await api.ensureDaemon(device);
|
|
53
|
+
} catch {
|
|
54
|
+
console.log('the device is not producing frames; reviving before starting');
|
|
55
|
+
try {
|
|
56
|
+
execFileSync(process.execPath, ['src/cli.js', 'revive', `--device=${device}`],
|
|
57
|
+
{ cwd: new URL('..', import.meta.url).pathname, timeout: 300_000, stdio: 'inherit' });
|
|
58
|
+
} catch { /* reported below by the throw from ensureDaemon */ }
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
const { device: dev } = await api.ensureDaemon(device);
|
|
63
|
+
console.log(`device: ${dev.name} (${dev.runtime})`);
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Cold-launch the app with a seed.
|
|
67
|
+
*
|
|
68
|
+
* `openUrl` on a terminated app launches it with that URL as its initial URL,
|
|
69
|
+
* which is the only way to set the seed *before* the first screen mounts and
|
|
70
|
+
* starts its own timers. Relaunching and then opening the URL would be too
|
|
71
|
+
* late: the list's delay has already been drawn from the default stream.
|
|
72
|
+
*/
|
|
73
|
+
/**
|
|
74
|
+
* Scaffolding runs with the supervisor OFF, and that is not a detail.
|
|
75
|
+
*
|
|
76
|
+
* The first collection run produced 18 rulings of which **14 came from the
|
|
77
|
+
* harness's own plumbing** — seven from tapping an "Open in …?" dialog that was
|
|
78
|
+
* not always there, two from terminating an app that was not running, three
|
|
79
|
+
* from typing into a form we had failed to reach. Every one was a real
|
|
80
|
+
* consultation and every one would have landed in the population that 101 and
|
|
81
|
+
* 96 are going to measure.
|
|
82
|
+
*
|
|
83
|
+
* A fixture is a claim about what the supervisor should say. Plumbing is not,
|
|
84
|
+
* and a harness that cannot tell them apart is measuring itself.
|
|
85
|
+
*/
|
|
86
|
+
const SCAFFOLD = { supervisor: 'none' };
|
|
87
|
+
|
|
88
|
+
const launchSeeded = async (seed) => {
|
|
89
|
+
// Terminating an app that is not running fails, and a failed step aborts the
|
|
90
|
+
// rest of the batch — so it gets its own call and its own shrug.
|
|
91
|
+
await actions.runScript(device, { steps: [{ terminate: BUNDLE }], verify: false, options: SCAFFOLD }).catch(() => null);
|
|
92
|
+
await actions.runScript(device, {
|
|
93
|
+
steps: [{ openUrl: `simframetestbed://seed/${seed}` }, { pause: 1200 }],
|
|
94
|
+
verify: false,
|
|
95
|
+
options: SCAFFOLD,
|
|
96
|
+
});
|
|
97
|
+
// iOS asks "Open in ...?" whenever a custom scheme is opened by another
|
|
98
|
+
// process, springboard included, and it asks on every launch. Answered here
|
|
99
|
+
// rather than designed around: the alternative is launch arguments, which RN
|
|
100
|
+
// does not expose to JavaScript without a native module.
|
|
101
|
+
await actions.runScript(device, {
|
|
102
|
+
steps: [{ tap: 'Open' }, { pause: 2200 }],
|
|
103
|
+
verify: false,
|
|
104
|
+
options: SCAFFOLD,
|
|
105
|
+
}).catch(async () => {
|
|
106
|
+
// No dialog this time. Give the app the same settling time anyway, so the
|
|
107
|
+
// fixture's timing does not depend on whether iOS felt like asking.
|
|
108
|
+
await actions.runScript(device, { steps: [{ pause: 2200 }], verify: false, options: SCAFFOLD }).catch(() => null);
|
|
109
|
+
});
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Walk to where a fixture starts, unjudged — and *prove* you arrived.
|
|
114
|
+
*
|
|
115
|
+
* Navigation is scaffolding too. Three of the first run's stray rulings were a
|
|
116
|
+
* `type` that failed because the form had never been reached.
|
|
117
|
+
*
|
|
118
|
+
* The `arrive` half is the harder lesson, and it cost a wrong number. A walk
|
|
119
|
+
* that does not throw is not a walk that arrived: two `Next` taps landed on a
|
|
120
|
+
* live button, threw nothing, and advanced nothing, so **three of four rulings
|
|
121
|
+
* in the first clean run were taken on step 1 of a three-step form** while the
|
|
122
|
+
* fixture claimed they were about the review step. The scoreboard read 25% and
|
|
123
|
+
* was measuring the harness.
|
|
124
|
+
*
|
|
125
|
+
* `eval-fingerprint.mjs` already learned exactly this — it checks that each
|
|
126
|
+
* reading was taken on the screen the tour named, having once measured a
|
|
127
|
+
* distribution against readings taken somewhere else. The check simply had not
|
|
128
|
+
* been carried over.
|
|
129
|
+
*/
|
|
130
|
+
const walk = async (steps, arrive) => {
|
|
131
|
+
try {
|
|
132
|
+
if (steps.length) await actions.runScript(device, { steps, verify: false, options: SCAFFOLD });
|
|
133
|
+
if (!arrive) return true;
|
|
134
|
+
// Asserted, not assumed. An assert that throws means we are not there.
|
|
135
|
+
await actions.runScript(device, {
|
|
136
|
+
steps: [{ assert: { value: arrive, is: 'visible' } }],
|
|
137
|
+
verify: false,
|
|
138
|
+
options: SCAFFOLD,
|
|
139
|
+
});
|
|
140
|
+
return true;
|
|
141
|
+
} catch {
|
|
142
|
+
return false;
|
|
143
|
+
}
|
|
144
|
+
};
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Revive the device if capture has given up, and keep going.
|
|
148
|
+
*
|
|
149
|
+
* Collecting a population means driving one simulator hard for several minutes,
|
|
150
|
+
* and the display stops rendering when you do — twice in one afternoon here. The
|
|
151
|
+
* daemon detects it, tries both its recoveries, reports `stalled` and stops,
|
|
152
|
+
* because a capture loop that rebooted the device it was watching would be a
|
|
153
|
+
* tool reaching for the mains. This is a harness, not the product, and the
|
|
154
|
+
* operator's answer is exactly what it is here to automate — otherwise a run of
|
|
155
|
+
* twenty fixtures ends at the third and the population is however many rulings
|
|
156
|
+
* the device survived.
|
|
157
|
+
*/
|
|
158
|
+
const reviveIfWedged = async () => {
|
|
159
|
+
const health = store.captureHealth(dev.udid);
|
|
160
|
+
if (!health?.stalled) return false;
|
|
161
|
+
process.stdout.write(' (capture stalled — reviving the device before continuing)\n');
|
|
162
|
+
await new Promise((resolve) => {
|
|
163
|
+
execFile(process.execPath, ['src/cli.js', 'revive', `--device=${dev.udid}`],
|
|
164
|
+
{ timeout: 300_000, cwd: new URL('..', import.meta.url).pathname }, () => resolve());
|
|
165
|
+
});
|
|
166
|
+
return true;
|
|
167
|
+
};
|
|
168
|
+
|
|
169
|
+
const FIXTURES = [
|
|
170
|
+
{
|
|
171
|
+
name: 'arriving',
|
|
172
|
+
want: 'recovered',
|
|
173
|
+
// Pull to refresh rather than relying on the launch, because reaching the
|
|
174
|
+
// fixture now takes three seconds of its own — answering iOS's "Open in …?"
|
|
175
|
+
// — and by then the list has always arrived. The first clean run produced
|
|
176
|
+
// *no rulings at all* from this fixture for that reason. A refresh empties
|
|
177
|
+
// the list and reloads it on a fresh seeded delay, right where we want it.
|
|
178
|
+
walk: [{ swipe: { from: [201, 300], to: [201, 620] } }],
|
|
179
|
+
arrive: null,
|
|
180
|
+
judge: { waitFor: { value: 'Monstera #1', timeoutMs: 700 } },
|
|
181
|
+
expect: 'the list is still loading; its rows arrive shortly after launch',
|
|
182
|
+
},
|
|
183
|
+
{
|
|
184
|
+
name: 'detail',
|
|
185
|
+
want: 'recovered',
|
|
186
|
+
// A detail screen that is still fetching. Waiting is the right answer and
|
|
187
|
+
// re-running the step proves it, which is what makes this scoreable.
|
|
188
|
+
walk: [{ tap: 'Monstera #1' }],
|
|
189
|
+
arrive: null,
|
|
190
|
+
judge: { waitFor: { value: 'Prefers bright indirect light', timeoutMs: 700 } },
|
|
191
|
+
expect: 'the detail screen is still fetching; its text arrives shortly',
|
|
192
|
+
},
|
|
193
|
+
{
|
|
194
|
+
name: 'secondwave',
|
|
195
|
+
want: 'recovered',
|
|
196
|
+
// The list renders its count header, then a third of its rows, then the
|
|
197
|
+
// rest. `Jade #24` is in the last wave, so a tight wait for it fails while
|
|
198
|
+
// the screen is *stable and incomplete at the same moment* — the state that
|
|
199
|
+
// has fooled settle detection and the supervisor alike.
|
|
200
|
+
walk: [{ swipe: { from: [201, 300], to: [201, 620] } }],
|
|
201
|
+
arrive: null,
|
|
202
|
+
judge: { waitFor: { value: 'Jade #24', timeoutMs: 700 } },
|
|
203
|
+
expect: 'the list arrives in waves and this row is in the last one',
|
|
204
|
+
},
|
|
205
|
+
{
|
|
206
|
+
name: 'blocked',
|
|
207
|
+
want: 'stopped',
|
|
208
|
+
walk: [
|
|
209
|
+
{ tap: 'Forms, tab, 2 of 3' }, { pause: 900 }, { tap: 'Stepped form' }, { pause: 900 },
|
|
210
|
+
{ tap: 'Next' }, { pause: 1200 }, { tap: 'Next' }, { pause: 1200 },
|
|
211
|
+
],
|
|
212
|
+
// The review step, proved rather than hoped for.
|
|
213
|
+
arrive: 'Step 3 of 3',
|
|
214
|
+
judge: { waitFor: { value: 'Submitted', timeoutMs: 2500 } },
|
|
215
|
+
expect: 'Review is blocked until Species is filled in, and it is empty',
|
|
216
|
+
},
|
|
217
|
+
{
|
|
218
|
+
name: 'refused',
|
|
219
|
+
want: 'stopped',
|
|
220
|
+
walk: [
|
|
221
|
+
{ tap: 'Forms, tab, 2 of 3' }, { pause: 900 }, { tap: 'One-step form' },
|
|
222
|
+
{ pause: 6500 },
|
|
223
|
+
{ type: { into: 'Your Name', text: 'Ada' } },
|
|
224
|
+
{ tap: 'Submit' }, { pause: 1200 },
|
|
225
|
+
],
|
|
226
|
+
// The rejection is on screen, so the submit demonstrably happened and
|
|
227
|
+
// failed — otherwise this fixture can pass by never having submitted.
|
|
228
|
+
arrive: 'The order was rejected',
|
|
229
|
+
// "Saved" was the string here and it fuzzy-matched "Could not save: …", so
|
|
230
|
+
// the judge step *succeeded* on the failure it was meant to catch and the
|
|
231
|
+
// fixture produced no rulings at all. Success and failure now share no words.
|
|
232
|
+
judge: { waitFor: { value: 'Order placed', timeoutMs: 2500 } },
|
|
233
|
+
expect: 'the first submit always fails and the second works, so waiting cannot help',
|
|
234
|
+
},
|
|
235
|
+
];
|
|
236
|
+
|
|
237
|
+
const before = metrics.readSupervisions(dev.udid).length;
|
|
238
|
+
let runs = 0;
|
|
239
|
+
let skipped = 0;
|
|
240
|
+
|
|
241
|
+
for (let i = 0; i < seeds; i += 1) {
|
|
242
|
+
const seed = 1000 + i * 7;
|
|
243
|
+
for (const fx of FIXTURES) {
|
|
244
|
+
await reviveIfWedged();
|
|
245
|
+
await launchSeeded(seed);
|
|
246
|
+
const reached = await walk(fx.walk, fx.arrive);
|
|
247
|
+
if (!reached) {
|
|
248
|
+
process.stdout.write(` seed ${seed} ${fx.name.padEnd(9)} SKIPPED — could not reach the fixture\n`);
|
|
249
|
+
skipped += 1;
|
|
250
|
+
continue;
|
|
251
|
+
}
|
|
252
|
+
try {
|
|
253
|
+
await actions.runScript(device, {
|
|
254
|
+
steps: [{ ...fx.judge, expect: fx.expect }],
|
|
255
|
+
supervise: fx.name,
|
|
256
|
+
verify: true,
|
|
257
|
+
});
|
|
258
|
+
} catch { /* a failing step is the point */ }
|
|
259
|
+
runs += 1;
|
|
260
|
+
process.stdout.write(` seed ${seed} ${fx.name.padEnd(9)} judged\n`);
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
const all = metrics.readSupervisions(dev.udid);
|
|
265
|
+
const fresh = all.slice(before);
|
|
266
|
+
console.log(`\n${runs} judged step(s), ${skipped} skipped, ${fresh.length} ruling(s)\n`);
|
|
267
|
+
|
|
268
|
+
const byFixture = new Map(FIXTURES.map((f) => [f.expect, f]));
|
|
269
|
+
const score = new Map(FIXTURES.map((f) => [f.name, { n: 0, right: 0, decisions: {}, outcomes: {} }]));
|
|
270
|
+
let unattributed = 0;
|
|
271
|
+
for (const r of fresh) {
|
|
272
|
+
const fx = byFixture.get(r.expect);
|
|
273
|
+
if (!fx) { unattributed += 1; continue; }
|
|
274
|
+
const s = score.get(fx.name);
|
|
275
|
+
s.n += 1;
|
|
276
|
+
s.decisions[r.decision] = (s.decisions[r.decision] ?? 0) + 1;
|
|
277
|
+
s.outcomes[r.outcome] = (s.outcomes[r.outcome] ?? 0) + 1;
|
|
278
|
+
if (r.outcome === fx.want) s.right += 1;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
console.log(`${'fixture'.padEnd(10)} ${'n'.padStart(3)} ${'right'.padStart(6)} wanted decisions / outcomes`);
|
|
282
|
+
for (const fx of FIXTURES) {
|
|
283
|
+
const s = score.get(fx.name);
|
|
284
|
+
const pct = s.n ? `${Math.round((100 * s.right) / s.n)}%` : '—';
|
|
285
|
+
console.log(
|
|
286
|
+
`${fx.name.padEnd(10)} ${String(s.n).padStart(3)} ${pct.padStart(6)} ${fx.want.padEnd(10)} `
|
|
287
|
+
+ `${JSON.stringify(s.decisions)} / ${JSON.stringify(s.outcomes)}`,
|
|
288
|
+
);
|
|
289
|
+
}
|
|
290
|
+
if (unattributed) console.log(`\n${unattributed} ruling(s) came from a step this script did not label.`);
|
|
291
|
+
|
|
292
|
+
// Said before the accuracy is read, not after. The first population was 12
|
|
293
|
+
// `stop` to 2 `wait`, so a majority-class guess scored 86% against the model's
|
|
294
|
+
// 64% — and every number computed from it was an artifact of that skew. A
|
|
295
|
+
// scoreboard that prints accuracy without printing its own balance invites
|
|
296
|
+
// exactly that mistake a second time.
|
|
297
|
+
{
|
|
298
|
+
const want = {};
|
|
299
|
+
for (const fx of FIXTURES) {
|
|
300
|
+
const s = score.get(fx.name);
|
|
301
|
+
want[fx.want] = (want[fx.want] ?? 0) + s.n;
|
|
302
|
+
}
|
|
303
|
+
const total = Object.values(want).reduce((a, b) => a + b, 0);
|
|
304
|
+
const biggest = Math.max(0, ...Object.values(want));
|
|
305
|
+
const baseline = total ? Math.round((100 * biggest) / total) : 0;
|
|
306
|
+
console.log(`\nbalance: ${JSON.stringify(want)} — guessing the commonest answer scores ${baseline}%.`);
|
|
307
|
+
if (baseline > 65) {
|
|
308
|
+
console.log(' SKEWED. Any accuracy above is mostly a fact about the fixture set, not the judge.');
|
|
309
|
+
console.log(' Add fixtures for the under-represented answer before comparing anything against anything.');
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
console.log(`\nthe log now holds ${all.length} ruling(s) — simframe supervisions --device=${dev.udid}`);
|
|
@@ -87,6 +87,43 @@ const readings = [];
|
|
|
87
87
|
/** Navigations that did not land before the reading was taken. */
|
|
88
88
|
const arrivalFailures = [];
|
|
89
89
|
|
|
90
|
+
/**
|
|
91
|
+
* The tokens that carry a name, as opposed to a shape.
|
|
92
|
+
*
|
|
93
|
+
* A fingerprint is deliberately geometry — role, region, size, position — and
|
|
94
|
+
* chrome labels are the only text that survives into it (`fingerprint.js`).
|
|
95
|
+
* That module's own comment states the consequence: "two list screens with
|
|
96
|
+
* identical structure differ by their title, and nothing else says so". So a
|
|
97
|
+
* reading with none of these has no identity to speak of, and two such
|
|
98
|
+
* readings of *different* screens can hash identically. Counted here because
|
|
99
|
+
* that is diagnosable and "the tour went somewhere unintended" is not.
|
|
100
|
+
*/
|
|
101
|
+
const namedTokens = (tokens) => tokens.filter((t) => t.includes('"'));
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Write the readings out now, rather than after the checks.
|
|
105
|
+
*
|
|
106
|
+
* The write used to sit past every `process.exit(1)`, so the only run that
|
|
107
|
+
* kept its evidence was the run with nothing to explain. A failing run exited
|
|
108
|
+
* before the file existed and `if: always()` on the upload step faithfully
|
|
109
|
+
* uploaded nothing — which is how one red integration job cost an evening of
|
|
110
|
+
* inferring from a summary line while `analyse-fingerprint.mjs`, which exists
|
|
111
|
+
* to classify exactly these divergences, had no file to read. Called as soon
|
|
112
|
+
* as the tour is done and again with the analysis, so every exit below this
|
|
113
|
+
* point still leaves the readings behind.
|
|
114
|
+
*/
|
|
115
|
+
const save = (extra = {}) => {
|
|
116
|
+
if (!outFile) return;
|
|
117
|
+
fs.writeFileSync(outFile, JSON.stringify({
|
|
118
|
+
label, device: dev.name, runtime: dev.runtime, rounds, at: Date.now(),
|
|
119
|
+
threshold: graph.SIMILARITY_THRESHOLD,
|
|
120
|
+
// Tokens are kept. They were stripped here once, and the first time the
|
|
121
|
+
// margin narrowed the run could not be diagnosed from its own output.
|
|
122
|
+
readings,
|
|
123
|
+
...extra,
|
|
124
|
+
}, null, 2));
|
|
125
|
+
};
|
|
126
|
+
|
|
90
127
|
for (let round = 1; round <= rounds; round += 1) {
|
|
91
128
|
for (const screen of tour) {
|
|
92
129
|
if (screen.steps?.length) {
|
|
@@ -136,12 +173,25 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
136
173
|
// like fingerprint drift.
|
|
137
174
|
sources: id.entry?.sources ?? [],
|
|
138
175
|
});
|
|
176
|
+
// Sensors and named tokens are on every line, not only inside a failure.
|
|
177
|
+
// Both were invisible until a run failed, and both are what the failure
|
|
178
|
+
// turns out to be about: two readings of one screen taken by different
|
|
179
|
+
// sensors do not share a hash by design, and a reading carrying no chrome
|
|
180
|
+
// label cannot be told from any other screen of the same shape. A summary
|
|
181
|
+
// line that hid those sent an evening after hosted-runner speed.
|
|
139
182
|
process.stdout.write(
|
|
140
|
-
` round ${round} ${screen.name.padEnd(
|
|
183
|
+
` round ${round} ${screen.name.padEnd(22)} ${String(id.hash).slice(0, 10)} `
|
|
184
|
+
+ `${String((id.tokens ?? []).length).padStart(3)} tokens `
|
|
185
|
+
+ `${String(namedTokens(id.tokens ?? []).length).padStart(2)} named `
|
|
186
|
+
+ `${((id.entry?.sources ?? []).join('+') || 'none').padEnd(9)}`
|
|
187
|
+
+ `${id.settled ? '' : ' (never settled)'}\n`,
|
|
141
188
|
);
|
|
142
189
|
}
|
|
143
190
|
}
|
|
144
191
|
|
|
192
|
+
save();
|
|
193
|
+
if (outFile) console.log(`\nwrote ${readings.length} readings to ${outFile}`);
|
|
194
|
+
|
|
145
195
|
if (arrivalFailures.length) {
|
|
146
196
|
console.error(`\nFAIL ${arrivalFailures.length} reading(s) were taken on the previous screen:`);
|
|
147
197
|
for (const f of arrivalFailures) console.error(` ${f}`);
|
|
@@ -178,11 +228,17 @@ function findStrays(all) {
|
|
|
178
228
|
if (siblings.length < 2) continue;
|
|
179
229
|
const bestSelf = Math.max(...siblings.map((o) => fingerprint.similarity(r.tokens, o.tokens)));
|
|
180
230
|
const others = all.filter((o) => o.name !== r.name);
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
231
|
+
// Which screen it resembles, not merely how much. A stray that resembles
|
|
232
|
+
// one particular other screen at 1.00 is a different animal from one that
|
|
233
|
+
// resembles everything weakly, and the report could not tell them apart.
|
|
234
|
+
let match = null;
|
|
235
|
+
let bestOther = 0;
|
|
236
|
+
for (const o of others) {
|
|
237
|
+
const s = fingerprint.similarity(r.tokens, o.tokens);
|
|
238
|
+
if (s > bestOther) { bestOther = s; match = o; }
|
|
239
|
+
}
|
|
184
240
|
if (bestSelf < graph.SIMILARITY_THRESHOLD && bestOther >= bestSelf) {
|
|
185
|
-
strays.push({ reading: r, bestSelf, bestOther });
|
|
241
|
+
strays.push({ reading: r, bestSelf, bestOther, match });
|
|
186
242
|
}
|
|
187
243
|
}
|
|
188
244
|
return strays;
|
|
@@ -190,14 +246,41 @@ function findStrays(all) {
|
|
|
190
246
|
|
|
191
247
|
const strays = findStrays(readings);
|
|
192
248
|
if (strays.length) {
|
|
193
|
-
console.error(`\nFAIL ${strays.length} reading(s)
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
249
|
+
console.error(`\nFAIL ${strays.length} reading(s) do not resemble their own screen:`);
|
|
250
|
+
// Two causes wear the same symptom, and until now the report asserted the
|
|
251
|
+
// second one. A reading can be unlike its siblings because the tour went
|
|
252
|
+
// somewhere unintended — or because the fingerprint could not tell two
|
|
253
|
+
// screens apart, which is the harness's actual subject. They are separable
|
|
254
|
+
// from the data in hand: a collision is a reading that carries no chrome
|
|
255
|
+
// label while matching one particular other screen almost exactly, and a
|
|
256
|
+
// wrong turn is one whose tokens name a screen the tour did not ask for.
|
|
257
|
+
let collisions = 0;
|
|
258
|
+
for (const { reading, bestSelf, bestOther, match } of strays) {
|
|
259
|
+
const named = namedTokens(reading.tokens);
|
|
260
|
+
const matchNamed = match ? namedTokens(match.tokens) : [];
|
|
261
|
+
const collided = bestOther >= 0.99 && named.length === 0 && matchNamed.length === 0;
|
|
262
|
+
if (collided) collisions += 1;
|
|
263
|
+
console.error(` ${reading.name} r${reading.round}: own screen ${bestSelf.toFixed(2)}, `
|
|
264
|
+
+ `${match ? `${match.name} r${match.round}` : 'another screen'} ${bestOther.toFixed(2)} `
|
|
265
|
+
+ `(${reading.count} tokens, ${named.length} named, sources ${reading.sources.join('+') || 'none'})`);
|
|
266
|
+
if (collided) {
|
|
267
|
+
console.error(' ^ a COLLISION, not a wrong turn: neither reading carries a chrome');
|
|
268
|
+
console.error(' label, so both are structure with no name and the fingerprint has');
|
|
269
|
+
console.error(' nothing left to tell two list screens apart.');
|
|
270
|
+
}
|
|
271
|
+
if (named.length) console.error(` names: ${named.map((t) => t.slice(t.indexOf('"'), t.lastIndexOf('"') + 1)).join(' ')}`);
|
|
272
|
+
}
|
|
273
|
+
if (collisions) {
|
|
274
|
+
console.error(`\n${collisions} of ${strays.length} are fingerprint collisions. That is this harness's own subject,`);
|
|
275
|
+
console.error('not a tour fault: a reading whose chrome label went missing cannot establish');
|
|
276
|
+
console.error('identity, and comparing it as though it could is what produced the verdict above.');
|
|
277
|
+
} else {
|
|
278
|
+
console.error('\nThat is the tour going somewhere unintended, not the fingerprint drifting, and');
|
|
279
|
+
console.error('measuring it as either distribution poisons both ends. Fix the tour — a tap that');
|
|
280
|
+
console.error('missed, or a screen that needs longer than its pause — and re-run.');
|
|
197
281
|
}
|
|
198
|
-
console.error(
|
|
199
|
-
|
|
200
|
-
console.error('missed, or a screen that needs longer than its pause — and re-run.');
|
|
282
|
+
console.error(`\nEvery reading is in ${outFile ?? 'the --out file'}; `
|
|
283
|
+
+ 'run `node scripts/analyse-fingerprint.mjs <that file>` to classify the divergent tokens.');
|
|
201
284
|
process.exit(1);
|
|
202
285
|
}
|
|
203
286
|
|
|
@@ -317,17 +400,11 @@ for (const r of readings) {
|
|
|
317
400
|
}
|
|
318
401
|
console.log(`\n${labels.size} distinct chrome label(s) entered identity: ${[...labels].sort().join(' · ') || '(none)'}`);
|
|
319
402
|
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
// Tokens are kept. They were stripped here, and the first time the margin
|
|
326
|
-
// narrowed the run could not be diagnosed from its own output.
|
|
327
|
-
readings,
|
|
328
|
-
}, null, 2));
|
|
329
|
-
console.log(`\nwrote ${outFile}`);
|
|
330
|
-
}
|
|
403
|
+
save({
|
|
404
|
+
same: s, mixed: m, different: d, gap, separated, thresholdInGap,
|
|
405
|
+
labels: [...labels].sort(),
|
|
406
|
+
});
|
|
407
|
+
if (outFile) console.log(`\nwrote ${outFile}`);
|
|
331
408
|
|
|
332
409
|
// The stated margin, checked rather than eyeballed. A person noticing that a
|
|
333
410
|
// number moved is not a test; this is the machine that re-measures it.
|