simframe 0.11.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +44 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +29 -1
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +20 -2
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +41 -6
- package/package.json +1 -1
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/probe-network.mjs +118 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +20 -2
- package/src/actions.js +108 -9
- package/src/cli.js +85 -5
- package/src/fingerprint.js +25 -1
- package/src/graph.js +47 -0
- package/src/index.js +41 -0
- package/src/localhelper.js +8 -2
- package/src/mcp.js +16 -7
- package/src/metrics.js +116 -0
- package/src/platform/android.js +23 -0
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +23 -0
- package/src/regions.js +105 -0
- package/src/supervisor.js +51 -7
- package/src/view.js +40 -4
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Generate a population of supervisor rulings, and score them.
|
|
3
|
+
//
|
|
4
|
+
// Items 101, 96, 106 and 109a all need rulings to replay, and the log held one,
|
|
5
|
+
// because a ruling requires a step that genuinely fails and Apple's own apps do
|
|
6
|
+
// not fail on command. The React Native testbed does, from a seeded stream, so
|
|
7
|
+
// this turns "we need dozens of rulings" into a script.
|
|
8
|
+
//
|
|
9
|
+
// Every fixture here has a **known correct outcome**, which is what makes the
|
|
10
|
+
// population scoreable rather than merely large:
|
|
11
|
+
//
|
|
12
|
+
// arriving a list still loading. Re-running the step works, so the right
|
|
13
|
+
// answer is wait/retry and the right outcome is `recovered`.
|
|
14
|
+
// blocked a required field is empty, so the thing waited for can never
|
|
15
|
+
// appear. Waiting and retrying are both wrong; `stopped` is right.
|
|
16
|
+
// refused a submit that failed. Re-running the *wait* cannot help — only
|
|
17
|
+
// re-submitting could, and that is the planner's call, not the
|
|
18
|
+
// supervisor's — so `stopped` is right here too.
|
|
19
|
+
//
|
|
20
|
+
// Note `refused` and `blocked` want the same answer for different reasons. That
|
|
21
|
+
// is deliberate: a judge that says stop for the wrong reason is still right, and
|
|
22
|
+
// a population that only contains obvious cases measures nothing.
|
|
23
|
+
//
|
|
24
|
+
// SIMFRAME_SUPERVISOR=apple node scripts/collect-rulings.mjs --device=<udid> --seeds=8
|
|
25
|
+
import { execFile } from 'node:child_process';
|
|
26
|
+
import * as actions from '../src/actions.js';
|
|
27
|
+
import * as api from '../src/index.js';
|
|
28
|
+
import * as metrics from '../src/metrics.js';
|
|
29
|
+
import * as store from '../src/store.js';
|
|
30
|
+
import * as supervisor from '../src/supervisor.js';
|
|
31
|
+
|
|
32
|
+
const arg = (name, fallback) => {
|
|
33
|
+
const hit = process.argv.find((a) => a.startsWith(`--${name}=`));
|
|
34
|
+
return hit ? hit.slice(name.length + 3) : fallback;
|
|
35
|
+
};
|
|
36
|
+
const device = arg('device');
|
|
37
|
+
const seeds = Number(arg('seeds', 6));
|
|
38
|
+
const BUNDLE = 'com.example.simframetestbed';
|
|
39
|
+
|
|
40
|
+
if (!supervisor.requested({})) {
|
|
41
|
+
console.error('SIMFRAME_SUPERVISOR is not set, so nothing would be judged and no ruling would be recorded.');
|
|
42
|
+
process.exit(2);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// Before anything else, because the device may already be wedged from the last
|
|
46
|
+
// run — `ensureDaemon` throws on a display that has stopped rendering, and it
|
|
47
|
+
// threw here on the third attempt of the afternoon, before the loop's own check
|
|
48
|
+
// could ever run. Driving one simulator hard for a few minutes is what does it.
|
|
49
|
+
{
|
|
50
|
+
const { execFileSync } = await import('node:child_process');
|
|
51
|
+
try {
|
|
52
|
+
await api.ensureDaemon(device);
|
|
53
|
+
} catch {
|
|
54
|
+
console.log('the device is not producing frames; reviving before starting');
|
|
55
|
+
try {
|
|
56
|
+
execFileSync(process.execPath, ['src/cli.js', 'revive', `--device=${device}`],
|
|
57
|
+
{ cwd: new URL('..', import.meta.url).pathname, timeout: 300_000, stdio: 'inherit' });
|
|
58
|
+
} catch { /* reported below by the throw from ensureDaemon */ }
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
const { device: dev } = await api.ensureDaemon(device);
|
|
63
|
+
console.log(`device: ${dev.name} (${dev.runtime})`);
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Cold-launch the app with a seed.
|
|
67
|
+
*
|
|
68
|
+
* `openUrl` on a terminated app launches it with that URL as its initial URL,
|
|
69
|
+
* which is the only way to set the seed *before* the first screen mounts and
|
|
70
|
+
* starts its own timers. Relaunching and then opening the URL would be too
|
|
71
|
+
* late: the list's delay has already been drawn from the default stream.
|
|
72
|
+
*/
|
|
73
|
+
/**
|
|
74
|
+
* Scaffolding runs with the supervisor OFF, and that is not a detail.
|
|
75
|
+
*
|
|
76
|
+
* The first collection run produced 18 rulings of which **14 came from the
|
|
77
|
+
* harness's own plumbing** — seven from tapping an "Open in …?" dialog that was
|
|
78
|
+
* not always there, two from terminating an app that was not running, three
|
|
79
|
+
* from typing into a form we had failed to reach. Every one was a real
|
|
80
|
+
* consultation and every one would have landed in the population that 101 and
|
|
81
|
+
* 96 are going to measure.
|
|
82
|
+
*
|
|
83
|
+
* A fixture is a claim about what the supervisor should say. Plumbing is not,
|
|
84
|
+
* and a harness that cannot tell them apart is measuring itself.
|
|
85
|
+
*/
|
|
86
|
+
const SCAFFOLD = { supervisor: 'none' };
|
|
87
|
+
|
|
88
|
+
const launchSeeded = async (seed) => {
|
|
89
|
+
// Terminating an app that is not running fails, and a failed step aborts the
|
|
90
|
+
// rest of the batch — so it gets its own call and its own shrug.
|
|
91
|
+
await actions.runScript(device, { steps: [{ terminate: BUNDLE }], verify: false, options: SCAFFOLD }).catch(() => null);
|
|
92
|
+
await actions.runScript(device, {
|
|
93
|
+
steps: [{ openUrl: `simframetestbed://seed/${seed}` }, { pause: 1200 }],
|
|
94
|
+
verify: false,
|
|
95
|
+
options: SCAFFOLD,
|
|
96
|
+
});
|
|
97
|
+
// iOS asks "Open in ...?" whenever a custom scheme is opened by another
|
|
98
|
+
// process, springboard included, and it asks on every launch. Answered here
|
|
99
|
+
// rather than designed around: the alternative is launch arguments, which RN
|
|
100
|
+
// does not expose to JavaScript without a native module.
|
|
101
|
+
await actions.runScript(device, {
|
|
102
|
+
steps: [{ tap: 'Open' }, { pause: 2200 }],
|
|
103
|
+
verify: false,
|
|
104
|
+
options: SCAFFOLD,
|
|
105
|
+
}).catch(async () => {
|
|
106
|
+
// No dialog this time. Give the app the same settling time anyway, so the
|
|
107
|
+
// fixture's timing does not depend on whether iOS felt like asking.
|
|
108
|
+
await actions.runScript(device, { steps: [{ pause: 2200 }], verify: false, options: SCAFFOLD }).catch(() => null);
|
|
109
|
+
});
|
|
110
|
+
};
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Walk to where a fixture starts, unjudged — and *prove* you arrived.
|
|
114
|
+
*
|
|
115
|
+
* Navigation is scaffolding too. Three of the first run's stray rulings were a
|
|
116
|
+
* `type` that failed because the form had never been reached.
|
|
117
|
+
*
|
|
118
|
+
* The `arrive` half is the harder lesson, and it cost a wrong number. A walk
|
|
119
|
+
* that does not throw is not a walk that arrived: two `Next` taps landed on a
|
|
120
|
+
* live button, threw nothing, and advanced nothing, so **three of four rulings
|
|
121
|
+
* in the first clean run were taken on step 1 of a three-step form** while the
|
|
122
|
+
* fixture claimed they were about the review step. The scoreboard read 25% and
|
|
123
|
+
* was measuring the harness.
|
|
124
|
+
*
|
|
125
|
+
* `eval-fingerprint.mjs` already learned exactly this — it checks that each
|
|
126
|
+
* reading was taken on the screen the tour named, having once measured a
|
|
127
|
+
* distribution against readings taken somewhere else. The check simply had not
|
|
128
|
+
* been carried over.
|
|
129
|
+
*/
|
|
130
|
+
const walk = async (steps, arrive) => {
|
|
131
|
+
try {
|
|
132
|
+
if (steps.length) await actions.runScript(device, { steps, verify: false, options: SCAFFOLD });
|
|
133
|
+
if (!arrive) return true;
|
|
134
|
+
// Asserted, not assumed. An assert that throws means we are not there.
|
|
135
|
+
await actions.runScript(device, {
|
|
136
|
+
steps: [{ assert: { value: arrive, is: 'visible' } }],
|
|
137
|
+
verify: false,
|
|
138
|
+
options: SCAFFOLD,
|
|
139
|
+
});
|
|
140
|
+
return true;
|
|
141
|
+
} catch {
|
|
142
|
+
return false;
|
|
143
|
+
}
|
|
144
|
+
};
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Revive the device if capture has given up, and keep going.
|
|
148
|
+
*
|
|
149
|
+
* Collecting a population means driving one simulator hard for several minutes,
|
|
150
|
+
* and the display stops rendering when you do — twice in one afternoon here. The
|
|
151
|
+
* daemon detects it, tries both its recoveries, reports `stalled` and stops,
|
|
152
|
+
* because a capture loop that rebooted the device it was watching would be a
|
|
153
|
+
* tool reaching for the mains. This is a harness, not the product, and the
|
|
154
|
+
* operator's answer is exactly what it is here to automate — otherwise a run of
|
|
155
|
+
* twenty fixtures ends at the third and the population is however many rulings
|
|
156
|
+
* the device survived.
|
|
157
|
+
*/
|
|
158
|
+
const reviveIfWedged = async () => {
|
|
159
|
+
const health = store.captureHealth(dev.udid);
|
|
160
|
+
if (!health?.stalled) return false;
|
|
161
|
+
process.stdout.write(' (capture stalled — reviving the device before continuing)\n');
|
|
162
|
+
await new Promise((resolve) => {
|
|
163
|
+
execFile(process.execPath, ['src/cli.js', 'revive', `--device=${dev.udid}`],
|
|
164
|
+
{ timeout: 300_000, cwd: new URL('..', import.meta.url).pathname }, () => resolve());
|
|
165
|
+
});
|
|
166
|
+
return true;
|
|
167
|
+
};
|
|
168
|
+
|
|
169
|
+
const FIXTURES = [
|
|
170
|
+
{
|
|
171
|
+
name: 'arriving',
|
|
172
|
+
want: 'recovered',
|
|
173
|
+
// Pull to refresh rather than relying on the launch, because reaching the
|
|
174
|
+
// fixture now takes three seconds of its own — answering iOS's "Open in …?"
|
|
175
|
+
// — and by then the list has always arrived. The first clean run produced
|
|
176
|
+
// *no rulings at all* from this fixture for that reason. A refresh empties
|
|
177
|
+
// the list and reloads it on a fresh seeded delay, right where we want it.
|
|
178
|
+
walk: [{ swipe: { from: [201, 300], to: [201, 620] } }],
|
|
179
|
+
arrive: null,
|
|
180
|
+
judge: { waitFor: { value: 'Monstera #1', timeoutMs: 700 } },
|
|
181
|
+
expect: 'the list is still loading; its rows arrive shortly after launch',
|
|
182
|
+
},
|
|
183
|
+
{
|
|
184
|
+
name: 'detail',
|
|
185
|
+
want: 'recovered',
|
|
186
|
+
// A detail screen that is still fetching. Waiting is the right answer and
|
|
187
|
+
// re-running the step proves it, which is what makes this scoreable.
|
|
188
|
+
walk: [{ tap: 'Monstera #1' }],
|
|
189
|
+
arrive: null,
|
|
190
|
+
judge: { waitFor: { value: 'Prefers bright indirect light', timeoutMs: 700 } },
|
|
191
|
+
expect: 'the detail screen is still fetching; its text arrives shortly',
|
|
192
|
+
},
|
|
193
|
+
{
|
|
194
|
+
name: 'secondwave',
|
|
195
|
+
want: 'recovered',
|
|
196
|
+
// The list renders its count header, then a third of its rows, then the
|
|
197
|
+
// rest. `Jade #24` is in the last wave, so a tight wait for it fails while
|
|
198
|
+
// the screen is *stable and incomplete at the same moment* — the state that
|
|
199
|
+
// has fooled settle detection and the supervisor alike.
|
|
200
|
+
walk: [{ swipe: { from: [201, 300], to: [201, 620] } }],
|
|
201
|
+
arrive: null,
|
|
202
|
+
judge: { waitFor: { value: 'Jade #24', timeoutMs: 700 } },
|
|
203
|
+
expect: 'the list arrives in waves and this row is in the last one',
|
|
204
|
+
},
|
|
205
|
+
{
|
|
206
|
+
name: 'blocked',
|
|
207
|
+
want: 'stopped',
|
|
208
|
+
walk: [
|
|
209
|
+
{ tap: 'Forms, tab, 2 of 3' }, { pause: 900 }, { tap: 'Stepped form' }, { pause: 900 },
|
|
210
|
+
{ tap: 'Next' }, { pause: 1200 }, { tap: 'Next' }, { pause: 1200 },
|
|
211
|
+
],
|
|
212
|
+
// The review step, proved rather than hoped for.
|
|
213
|
+
arrive: 'Step 3 of 3',
|
|
214
|
+
judge: { waitFor: { value: 'Submitted', timeoutMs: 2500 } },
|
|
215
|
+
expect: 'Review is blocked until Species is filled in, and it is empty',
|
|
216
|
+
},
|
|
217
|
+
{
|
|
218
|
+
name: 'refused',
|
|
219
|
+
want: 'stopped',
|
|
220
|
+
walk: [
|
|
221
|
+
{ tap: 'Forms, tab, 2 of 3' }, { pause: 900 }, { tap: 'One-step form' },
|
|
222
|
+
{ pause: 6500 },
|
|
223
|
+
{ type: { into: 'Your Name', text: 'Ada' } },
|
|
224
|
+
{ tap: 'Submit' }, { pause: 1200 },
|
|
225
|
+
],
|
|
226
|
+
// The rejection is on screen, so the submit demonstrably happened and
|
|
227
|
+
// failed — otherwise this fixture can pass by never having submitted.
|
|
228
|
+
arrive: 'The order was rejected',
|
|
229
|
+
// "Saved" was the string here and it fuzzy-matched "Could not save: …", so
|
|
230
|
+
// the judge step *succeeded* on the failure it was meant to catch and the
|
|
231
|
+
// fixture produced no rulings at all. Success and failure now share no words.
|
|
232
|
+
judge: { waitFor: { value: 'Order placed', timeoutMs: 2500 } },
|
|
233
|
+
expect: 'the first submit always fails and the second works, so waiting cannot help',
|
|
234
|
+
},
|
|
235
|
+
];
|
|
236
|
+
|
|
237
|
+
const before = metrics.readSupervisions(dev.udid).length;
|
|
238
|
+
let runs = 0;
|
|
239
|
+
let skipped = 0;
|
|
240
|
+
|
|
241
|
+
for (let i = 0; i < seeds; i += 1) {
|
|
242
|
+
const seed = 1000 + i * 7;
|
|
243
|
+
for (const fx of FIXTURES) {
|
|
244
|
+
await reviveIfWedged();
|
|
245
|
+
await launchSeeded(seed);
|
|
246
|
+
const reached = await walk(fx.walk, fx.arrive);
|
|
247
|
+
if (!reached) {
|
|
248
|
+
process.stdout.write(` seed ${seed} ${fx.name.padEnd(9)} SKIPPED — could not reach the fixture\n`);
|
|
249
|
+
skipped += 1;
|
|
250
|
+
continue;
|
|
251
|
+
}
|
|
252
|
+
try {
|
|
253
|
+
await actions.runScript(device, {
|
|
254
|
+
steps: [{ ...fx.judge, expect: fx.expect }],
|
|
255
|
+
supervise: fx.name,
|
|
256
|
+
verify: true,
|
|
257
|
+
});
|
|
258
|
+
} catch { /* a failing step is the point */ }
|
|
259
|
+
runs += 1;
|
|
260
|
+
process.stdout.write(` seed ${seed} ${fx.name.padEnd(9)} judged\n`);
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
const all = metrics.readSupervisions(dev.udid);
|
|
265
|
+
const fresh = all.slice(before);
|
|
266
|
+
console.log(`\n${runs} judged step(s), ${skipped} skipped, ${fresh.length} ruling(s)\n`);
|
|
267
|
+
|
|
268
|
+
const byFixture = new Map(FIXTURES.map((f) => [f.expect, f]));
|
|
269
|
+
const score = new Map(FIXTURES.map((f) => [f.name, { n: 0, right: 0, decisions: {}, outcomes: {} }]));
|
|
270
|
+
let unattributed = 0;
|
|
271
|
+
for (const r of fresh) {
|
|
272
|
+
const fx = byFixture.get(r.expect);
|
|
273
|
+
if (!fx) { unattributed += 1; continue; }
|
|
274
|
+
const s = score.get(fx.name);
|
|
275
|
+
s.n += 1;
|
|
276
|
+
s.decisions[r.decision] = (s.decisions[r.decision] ?? 0) + 1;
|
|
277
|
+
s.outcomes[r.outcome] = (s.outcomes[r.outcome] ?? 0) + 1;
|
|
278
|
+
if (r.outcome === fx.want) s.right += 1;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
console.log(`${'fixture'.padEnd(10)} ${'n'.padStart(3)} ${'right'.padStart(6)} wanted decisions / outcomes`);
|
|
282
|
+
for (const fx of FIXTURES) {
|
|
283
|
+
const s = score.get(fx.name);
|
|
284
|
+
const pct = s.n ? `${Math.round((100 * s.right) / s.n)}%` : '—';
|
|
285
|
+
console.log(
|
|
286
|
+
`${fx.name.padEnd(10)} ${String(s.n).padStart(3)} ${pct.padStart(6)} ${fx.want.padEnd(10)} `
|
|
287
|
+
+ `${JSON.stringify(s.decisions)} / ${JSON.stringify(s.outcomes)}`,
|
|
288
|
+
);
|
|
289
|
+
}
|
|
290
|
+
if (unattributed) console.log(`\n${unattributed} ruling(s) came from a step this script did not label.`);
|
|
291
|
+
|
|
292
|
+
// Said before the accuracy is read, not after. The first population was 12
|
|
293
|
+
// `stop` to 2 `wait`, so a majority-class guess scored 86% against the model's
|
|
294
|
+
// 64% — and every number computed from it was an artifact of that skew. A
|
|
295
|
+
// scoreboard that prints accuracy without printing its own balance invites
|
|
296
|
+
// exactly that mistake a second time.
|
|
297
|
+
{
|
|
298
|
+
const want = {};
|
|
299
|
+
for (const fx of FIXTURES) {
|
|
300
|
+
const s = score.get(fx.name);
|
|
301
|
+
want[fx.want] = (want[fx.want] ?? 0) + s.n;
|
|
302
|
+
}
|
|
303
|
+
const total = Object.values(want).reduce((a, b) => a + b, 0);
|
|
304
|
+
const biggest = Math.max(0, ...Object.values(want));
|
|
305
|
+
const baseline = total ? Math.round((100 * biggest) / total) : 0;
|
|
306
|
+
console.log(`\nbalance: ${JSON.stringify(want)} — guessing the commonest answer scores ${baseline}%.`);
|
|
307
|
+
if (baseline > 65) {
|
|
308
|
+
console.log(' SKEWED. Any accuracy above is mostly a fact about the fixture set, not the judge.');
|
|
309
|
+
console.log(' Add fixtures for the under-represented answer before comparing anything against anything.');
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
console.log(`\nthe log now holds ${all.length} ruling(s) — simframe supervisions --device=${dev.udid}`);
|
|
@@ -87,6 +87,43 @@ const readings = [];
|
|
|
87
87
|
/** Navigations that did not land before the reading was taken. */
|
|
88
88
|
const arrivalFailures = [];
|
|
89
89
|
|
|
90
|
+
/**
|
|
91
|
+
* The tokens that carry a name, as opposed to a shape.
|
|
92
|
+
*
|
|
93
|
+
* A fingerprint is deliberately geometry — role, region, size, position — and
|
|
94
|
+
* chrome labels are the only text that survives into it (`fingerprint.js`).
|
|
95
|
+
* That module's own comment states the consequence: "two list screens with
|
|
96
|
+
* identical structure differ by their title, and nothing else says so". So a
|
|
97
|
+
* reading with none of these has no identity to speak of, and two such
|
|
98
|
+
* readings of *different* screens can hash identically. Counted here because
|
|
99
|
+
* that is diagnosable and "the tour went somewhere unintended" is not.
|
|
100
|
+
*/
|
|
101
|
+
const namedTokens = (tokens) => tokens.filter((t) => t.includes('"'));
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Write the readings out now, rather than after the checks.
|
|
105
|
+
*
|
|
106
|
+
* The write used to sit past every `process.exit(1)`, so the only run that
|
|
107
|
+
* kept its evidence was the run with nothing to explain. A failing run exited
|
|
108
|
+
* before the file existed and `if: always()` on the upload step faithfully
|
|
109
|
+
* uploaded nothing — which is how one red integration job cost an evening of
|
|
110
|
+
* inferring from a summary line while `analyse-fingerprint.mjs`, which exists
|
|
111
|
+
* to classify exactly these divergences, had no file to read. Called as soon
|
|
112
|
+
* as the tour is done and again with the analysis, so every exit below this
|
|
113
|
+
* point still leaves the readings behind.
|
|
114
|
+
*/
|
|
115
|
+
const save = (extra = {}) => {
|
|
116
|
+
if (!outFile) return;
|
|
117
|
+
fs.writeFileSync(outFile, JSON.stringify({
|
|
118
|
+
label, device: dev.name, runtime: dev.runtime, rounds, at: Date.now(),
|
|
119
|
+
threshold: graph.SIMILARITY_THRESHOLD,
|
|
120
|
+
// Tokens are kept. They were stripped here once, and the first time the
|
|
121
|
+
// margin narrowed the run could not be diagnosed from its own output.
|
|
122
|
+
readings,
|
|
123
|
+
...extra,
|
|
124
|
+
}, null, 2));
|
|
125
|
+
};
|
|
126
|
+
|
|
90
127
|
for (let round = 1; round <= rounds; round += 1) {
|
|
91
128
|
for (const screen of tour) {
|
|
92
129
|
if (screen.steps?.length) {
|
|
@@ -136,12 +173,25 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
136
173
|
// like fingerprint drift.
|
|
137
174
|
sources: id.entry?.sources ?? [],
|
|
138
175
|
});
|
|
176
|
+
// Sensors and named tokens are on every line, not only inside a failure.
|
|
177
|
+
// Both were invisible until a run failed, and both are what the failure
|
|
178
|
+
// turns out to be about: two readings of one screen taken by different
|
|
179
|
+
// sensors do not share a hash by design, and a reading carrying no chrome
|
|
180
|
+
// label cannot be told from any other screen of the same shape. A summary
|
|
181
|
+
// line that hid those sent an evening after hosted-runner speed.
|
|
139
182
|
process.stdout.write(
|
|
140
|
-
` round ${round} ${screen.name.padEnd(
|
|
183
|
+
` round ${round} ${screen.name.padEnd(22)} ${String(id.hash).slice(0, 10)} `
|
|
184
|
+
+ `${String((id.tokens ?? []).length).padStart(3)} tokens `
|
|
185
|
+
+ `${String(namedTokens(id.tokens ?? []).length).padStart(2)} named `
|
|
186
|
+
+ `${((id.entry?.sources ?? []).join('+') || 'none').padEnd(9)}`
|
|
187
|
+
+ `${id.settled ? '' : ' (never settled)'}\n`,
|
|
141
188
|
);
|
|
142
189
|
}
|
|
143
190
|
}
|
|
144
191
|
|
|
192
|
+
save();
|
|
193
|
+
if (outFile) console.log(`\nwrote ${readings.length} readings to ${outFile}`);
|
|
194
|
+
|
|
145
195
|
if (arrivalFailures.length) {
|
|
146
196
|
console.error(`\nFAIL ${arrivalFailures.length} reading(s) were taken on the previous screen:`);
|
|
147
197
|
for (const f of arrivalFailures) console.error(` ${f}`);
|
|
@@ -178,11 +228,17 @@ function findStrays(all) {
|
|
|
178
228
|
if (siblings.length < 2) continue;
|
|
179
229
|
const bestSelf = Math.max(...siblings.map((o) => fingerprint.similarity(r.tokens, o.tokens)));
|
|
180
230
|
const others = all.filter((o) => o.name !== r.name);
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
231
|
+
// Which screen it resembles, not merely how much. A stray that resembles
|
|
232
|
+
// one particular other screen at 1.00 is a different animal from one that
|
|
233
|
+
// resembles everything weakly, and the report could not tell them apart.
|
|
234
|
+
let match = null;
|
|
235
|
+
let bestOther = 0;
|
|
236
|
+
for (const o of others) {
|
|
237
|
+
const s = fingerprint.similarity(r.tokens, o.tokens);
|
|
238
|
+
if (s > bestOther) { bestOther = s; match = o; }
|
|
239
|
+
}
|
|
184
240
|
if (bestSelf < graph.SIMILARITY_THRESHOLD && bestOther >= bestSelf) {
|
|
185
|
-
strays.push({ reading: r, bestSelf, bestOther });
|
|
241
|
+
strays.push({ reading: r, bestSelf, bestOther, match });
|
|
186
242
|
}
|
|
187
243
|
}
|
|
188
244
|
return strays;
|
|
@@ -190,14 +246,41 @@ function findStrays(all) {
|
|
|
190
246
|
|
|
191
247
|
const strays = findStrays(readings);
|
|
192
248
|
if (strays.length) {
|
|
193
|
-
console.error(`\nFAIL ${strays.length} reading(s)
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
249
|
+
console.error(`\nFAIL ${strays.length} reading(s) do not resemble their own screen:`);
|
|
250
|
+
// Two causes wear the same symptom, and until now the report asserted the
|
|
251
|
+
// second one. A reading can be unlike its siblings because the tour went
|
|
252
|
+
// somewhere unintended — or because the fingerprint could not tell two
|
|
253
|
+
// screens apart, which is the harness's actual subject. They are separable
|
|
254
|
+
// from the data in hand: a collision is a reading that carries no chrome
|
|
255
|
+
// label while matching one particular other screen almost exactly, and a
|
|
256
|
+
// wrong turn is one whose tokens name a screen the tour did not ask for.
|
|
257
|
+
let collisions = 0;
|
|
258
|
+
for (const { reading, bestSelf, bestOther, match } of strays) {
|
|
259
|
+
const named = namedTokens(reading.tokens);
|
|
260
|
+
const matchNamed = match ? namedTokens(match.tokens) : [];
|
|
261
|
+
const collided = bestOther >= 0.99 && named.length === 0 && matchNamed.length === 0;
|
|
262
|
+
if (collided) collisions += 1;
|
|
263
|
+
console.error(` ${reading.name} r${reading.round}: own screen ${bestSelf.toFixed(2)}, `
|
|
264
|
+
+ `${match ? `${match.name} r${match.round}` : 'another screen'} ${bestOther.toFixed(2)} `
|
|
265
|
+
+ `(${reading.count} tokens, ${named.length} named, sources ${reading.sources.join('+') || 'none'})`);
|
|
266
|
+
if (collided) {
|
|
267
|
+
console.error(' ^ a COLLISION, not a wrong turn: neither reading carries a chrome');
|
|
268
|
+
console.error(' label, so both are structure with no name and the fingerprint has');
|
|
269
|
+
console.error(' nothing left to tell two list screens apart.');
|
|
270
|
+
}
|
|
271
|
+
if (named.length) console.error(` names: ${named.map((t) => t.slice(t.indexOf('"'), t.lastIndexOf('"') + 1)).join(' ')}`);
|
|
272
|
+
}
|
|
273
|
+
if (collisions) {
|
|
274
|
+
console.error(`\n${collisions} of ${strays.length} are fingerprint collisions. That is this harness's own subject,`);
|
|
275
|
+
console.error('not a tour fault: a reading whose chrome label went missing cannot establish');
|
|
276
|
+
console.error('identity, and comparing it as though it could is what produced the verdict above.');
|
|
277
|
+
} else {
|
|
278
|
+
console.error('\nThat is the tour going somewhere unintended, not the fingerprint drifting, and');
|
|
279
|
+
console.error('measuring it as either distribution poisons both ends. Fix the tour — a tap that');
|
|
280
|
+
console.error('missed, or a screen that needs longer than its pause — and re-run.');
|
|
197
281
|
}
|
|
198
|
-
console.error(
|
|
199
|
-
|
|
200
|
-
console.error('missed, or a screen that needs longer than its pause — and re-run.');
|
|
282
|
+
console.error(`\nEvery reading is in ${outFile ?? 'the --out file'}; `
|
|
283
|
+
+ 'run `node scripts/analyse-fingerprint.mjs <that file>` to classify the divergent tokens.');
|
|
201
284
|
process.exit(1);
|
|
202
285
|
}
|
|
203
286
|
|
|
@@ -317,17 +400,11 @@ for (const r of readings) {
|
|
|
317
400
|
}
|
|
318
401
|
console.log(`\n${labels.size} distinct chrome label(s) entered identity: ${[...labels].sort().join(' · ') || '(none)'}`);
|
|
319
402
|
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
// Tokens are kept. They were stripped here, and the first time the margin
|
|
326
|
-
// narrowed the run could not be diagnosed from its own output.
|
|
327
|
-
readings,
|
|
328
|
-
}, null, 2));
|
|
329
|
-
console.log(`\nwrote ${outFile}`);
|
|
330
|
-
}
|
|
403
|
+
save({
|
|
404
|
+
same: s, mixed: m, different: d, gap, separated, thresholdInGap,
|
|
405
|
+
labels: [...labels].sort(),
|
|
406
|
+
});
|
|
407
|
+
if (outFile) console.log(`\nwrote ${outFile}`);
|
|
331
408
|
|
|
332
409
|
// The stated margin, checked rather than eyeballed. A person noticing that a
|
|
333
410
|
// number moved is not a test; this is the machine that re-measures it.
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
// Minimal CDP client over a hand-rolled WebSocket — no dependencies, because
|
|
2
|
+
// simframe has exactly one runtime dependency and this would not be it.
|
|
3
|
+
//
|
|
4
|
+
// Node 20 has no global WebSocket, so the upgrade and the framing are done by
|
|
5
|
+
// hand. Item 80 records that the upgrade 401s without an `Origin` matching the
|
|
6
|
+
// inspector's host; this sends one.
|
|
7
|
+
import http from 'node:http';
|
|
8
|
+
import crypto from 'node:crypto';
|
|
9
|
+
|
|
10
|
+
const wsUrl = process.argv[2];
|
|
11
|
+
const seconds = Number(process.argv[3] ?? 20);
|
|
12
|
+
const u = new URL(wsUrl);
|
|
13
|
+
|
|
14
|
+
const key = crypto.randomBytes(16).toString('base64');
|
|
15
|
+
const req = http.request({
|
|
16
|
+
hostname: u.hostname,
|
|
17
|
+
port: u.port,
|
|
18
|
+
path: u.pathname + u.search,
|
|
19
|
+
headers: {
|
|
20
|
+
Connection: 'Upgrade',
|
|
21
|
+
Upgrade: 'websocket',
|
|
22
|
+
'Sec-WebSocket-Key': key,
|
|
23
|
+
'Sec-WebSocket-Version': '13',
|
|
24
|
+
Origin: `http://${u.host}`,
|
|
25
|
+
},
|
|
26
|
+
});
|
|
27
|
+
|
|
28
|
+
/** A client frame must be masked; the server's are not. */
|
|
29
|
+
const frame = (text) => {
|
|
30
|
+
const payload = Buffer.from(text, 'utf8');
|
|
31
|
+
const mask = crypto.randomBytes(4);
|
|
32
|
+
const head = payload.length < 126
|
|
33
|
+
? Buffer.from([0x81, 0x80 | payload.length])
|
|
34
|
+
: Buffer.concat([Buffer.from([0x81, 0xfe]), (() => { const b = Buffer.alloc(2); b.writeUInt16BE(payload.length); return b; })()]);
|
|
35
|
+
const masked = Buffer.from(payload);
|
|
36
|
+
for (let i = 0; i < masked.length; i += 1) masked[i] ^= mask[i % 4];
|
|
37
|
+
return Buffer.concat([head, mask, masked]);
|
|
38
|
+
};
|
|
39
|
+
|
|
40
|
+
req.on('upgrade', (res, socket) => {
|
|
41
|
+
console.log(`upgraded: ${res.statusCode} ${res.headers.upgrade ?? ''}`);
|
|
42
|
+
let buf = Buffer.alloc(0);
|
|
43
|
+
let id = 0;
|
|
44
|
+
const send = (method, params) => {
|
|
45
|
+
id += 1;
|
|
46
|
+
socket.write(frame(JSON.stringify({ id, method, params })));
|
|
47
|
+
console.log(`-> ${method}`);
|
|
48
|
+
return id;
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
const seen = new Map();
|
|
52
|
+
socket.on('data', (chunk) => {
|
|
53
|
+
buf = Buffer.concat([buf, chunk]);
|
|
54
|
+
// Enough framing to read small unfragmented text frames, which is all CDP
|
|
55
|
+
// sends here. Deliberately not a WebSocket implementation.
|
|
56
|
+
for (;;) {
|
|
57
|
+
if (buf.length < 2) return;
|
|
58
|
+
const len0 = buf[1] & 0x7f;
|
|
59
|
+
let offset = 2;
|
|
60
|
+
let len = len0;
|
|
61
|
+
if (len0 === 126) { if (buf.length < 4) return; len = buf.readUInt16BE(2); offset = 4; }
|
|
62
|
+
else if (len0 === 127) { if (buf.length < 10) return; len = Number(buf.readBigUInt64BE(2)); offset = 10; }
|
|
63
|
+
if (buf.length < offset + len) return;
|
|
64
|
+
const text = buf.slice(offset, offset + len).toString('utf8');
|
|
65
|
+
buf = buf.slice(offset + len);
|
|
66
|
+
try {
|
|
67
|
+
const msg = JSON.parse(text);
|
|
68
|
+
if (msg.method) {
|
|
69
|
+
seen.set(msg.method, (seen.get(msg.method) ?? 0) + 1);
|
|
70
|
+
// The contents are the whole question: "settled, and these three
|
|
71
|
+
// requests fired with these statuses" is only sayable if the status
|
|
72
|
+
// and the URL are actually in here.
|
|
73
|
+
if (msg.method.startsWith('Network.')) {
|
|
74
|
+
const p = msg.params ?? {};
|
|
75
|
+
const bits = [
|
|
76
|
+
p.requestId && `id=${p.requestId}`,
|
|
77
|
+
p.request?.method && `${p.request.method} ${p.request.url}`,
|
|
78
|
+
p.response && `status=${p.response.status} ${p.response.mimeType ?? ''}`,
|
|
79
|
+
p.response?.timing && 'timing=yes',
|
|
80
|
+
p.encodedDataLength != null && `bytes=${p.encodedDataLength}`,
|
|
81
|
+
p.type && `type=${p.type}`,
|
|
82
|
+
].filter(Boolean);
|
|
83
|
+
console.log(` ${msg.method}: ${bits.join(' ')}`);
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
else console.log(`<- reply #${msg.id} ${msg.error ? `ERROR ${JSON.stringify(msg.error)}` : JSON.stringify(msg.result).slice(0, 120)}`);
|
|
87
|
+
} catch { /* not JSON; ignore */ }
|
|
88
|
+
}
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
send('Network.enable', {});
|
|
92
|
+
send('Runtime.enable', {});
|
|
93
|
+
send('Log.enable', {});
|
|
94
|
+
|
|
95
|
+
// Trigger real traffic through the app's own JS runtime, rather than through
|
|
96
|
+
// its UI. This isolates the question — does React Native's networking layer
|
|
97
|
+
// report to CDP at all — from whether my app happens to call anything, and it
|
|
98
|
+
// needs no screen, so it does not fight whatever else is driving the device.
|
|
99
|
+
setTimeout(() => {
|
|
100
|
+
send('Runtime.evaluate', {
|
|
101
|
+
expression: "fetch('https://jsonplaceholder.typicode.com/todos/1')"
|
|
102
|
+
+ ".then(r => r.json()).then(j => 'FETCHED ' + JSON.stringify(j))",
|
|
103
|
+
awaitPromise: true,
|
|
104
|
+
returnByValue: true,
|
|
105
|
+
});
|
|
106
|
+
}, 1500);
|
|
107
|
+
|
|
108
|
+
setTimeout(() => {
|
|
109
|
+
console.log('\nevents received, by method:');
|
|
110
|
+
if (!seen.size) console.log(' (none)');
|
|
111
|
+
for (const [m, n] of [...seen].sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(4)}x ${m}`);
|
|
112
|
+
socket.destroy();
|
|
113
|
+
process.exit(0);
|
|
114
|
+
}, seconds * 1000);
|
|
115
|
+
});
|
|
116
|
+
req.on('response', (res) => { console.log(`NOT upgraded: ${res.statusCode}`); process.exit(1); });
|
|
117
|
+
req.on('error', (e) => { console.log('error:', e.message); process.exit(1); });
|
|
118
|
+
req.end();
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// How long can this device be driven before capture wedges?
|
|
3
|
+
//
|
|
4
|
+
// The wedge is the thing standing between us and a ruling population: the
|
|
5
|
+
// display stops rendering after a few minutes of hard driving, `simctl
|
|
6
|
+
// screenshot` fails too, both of the daemon's recoveries fail, and only a
|
|
7
|
+
// device restart cures it. Three times in one afternoon.
|
|
8
|
+
//
|
|
9
|
+
// A leak was found in our own code on 2026-09-12 — damage-callback registration
|
|
10
|
+
// was not idempotent, and the recovery loop called it on every attempt, so one
|
|
11
|
+
// log's 670 port re-resolves meant up to 670 live callbacks on one port, each
|
|
12
|
+
// invoked per redraw. That is a real leak with a plausible path to saturating
|
|
13
|
+
// the display service, and it is **not proof** that it is the cause. This is
|
|
14
|
+
// how we find out: drive until it dies, and report how long that took.
|
|
15
|
+
//
|
|
16
|
+
// node scripts/soak-capture.mjs --device=<udid> --minutes=25
|
|
17
|
+
//
|
|
18
|
+
// A number to compare against, not a pass/fail. Before the fix, the device
|
|
19
|
+
// wedged roughly every 10-20 minutes of this kind of work.
|
|
20
|
+
import * as actions from '../src/actions.js';
|
|
21
|
+
import * as api from '../src/index.js';
|
|
22
|
+
import * as store from '../src/store.js';
|
|
23
|
+
|
|
24
|
+
const arg = (n, d) => {
|
|
25
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
26
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
27
|
+
};
|
|
28
|
+
const device = arg('device');
|
|
29
|
+
const minutes = Number(arg('minutes', 20));
|
|
30
|
+
const BUNDLE = 'com.example.simframetestbed';
|
|
31
|
+
|
|
32
|
+
const { device: dev } = await api.ensureDaemon(device);
|
|
33
|
+
console.log(`device: ${dev.name} (${dev.runtime}) budget: ${minutes} minutes`);
|
|
34
|
+
|
|
35
|
+
const started = Date.now();
|
|
36
|
+
const deadline = started + minutes * 60_000;
|
|
37
|
+
let laps = 0;
|
|
38
|
+
let reads = 0;
|
|
39
|
+
const elapsed = () => ((Date.now() - started) / 60_000).toFixed(1);
|
|
40
|
+
|
|
41
|
+
// A lap is deliberately the kind of work that provokes it: launching, moving
|
|
42
|
+
// between screens, and a cold read on each — not idling with a poll.
|
|
43
|
+
const LAP = [
|
|
44
|
+
[{ launch: { value: BUNDLE, relaunch: true } }, { pause: 2500 }],
|
|
45
|
+
[{ tap: 'Forms, tab, 2 of 3' }, { pause: 800 }],
|
|
46
|
+
[{ tap: 'Long form' }, { pause: 900 }],
|
|
47
|
+
[{ scroll: 'down' }, { pause: 500 }],
|
|
48
|
+
[{ scroll: 'down' }, { pause: 500 }],
|
|
49
|
+
[{ tap: 'Plants, tab, 1 of 3' }, { pause: 900 }],
|
|
50
|
+
[{ tap: 'Diagnostics, tab, 3 of 3' }, { pause: 900 }],
|
|
51
|
+
];
|
|
52
|
+
|
|
53
|
+
while (Date.now() < deadline) {
|
|
54
|
+
for (const steps of LAP) {
|
|
55
|
+
try {
|
|
56
|
+
await actions.runScript(device, { steps, verify: false, options: { supervisor: 'none' } });
|
|
57
|
+
} catch { /* a failed step is not the subject; a dead display is */ }
|
|
58
|
+
try {
|
|
59
|
+
await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
60
|
+
reads += 1;
|
|
61
|
+
} catch (err) {
|
|
62
|
+
console.log(`\nWEDGED after ${elapsed()} minutes, ${laps} laps, ${reads} cold reads`);
|
|
63
|
+
console.log(` ${String(err.message).split('\n')[0]}`);
|
|
64
|
+
const health = store.captureHealth(dev.udid);
|
|
65
|
+
if (health) console.log(` captureHealth: ${JSON.stringify(health)}`);
|
|
66
|
+
process.exit(1);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
laps += 1;
|
|
70
|
+
if (laps % 5 === 0) process.stdout.write(` ${elapsed()}m ${laps} laps ${reads} reads still alive\n`);
|
|
71
|
+
}
|
|
72
|
+
console.log(`\nSURVIVED ${minutes} minutes: ${laps} laps, ${reads} cold reads, no wedge`);
|