simframe 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +128 -53
- package/native/simframed/Sources/PrivateAPI/AccessibilityBridge.swift +379 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +78 -4
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +32 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +24 -1
- package/native/simframed/Sources/SimframeCore/Element.swift +29 -1
- package/native/simframed/Sources/simframed/main.swift +137 -9
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +30 -0
- package/package.json +4 -2
- package/scripts/check-package.mjs +8 -0
- package/scripts/ci-memory.mjs +416 -0
- package/scripts/eval-fingerprint.mjs +192 -0
- package/scripts/sync-server-version.mjs +39 -0
- package/skills/simframe/SKILL.md +173 -0
- package/src/actions.js +189 -20
- package/src/cli.js +272 -116
- package/src/control.js +1 -0
- package/src/fingerprint.js +33 -0
- package/src/graph.js +3 -1
- package/src/index.js +62 -4
- package/src/input.js +99 -0
- package/src/matching.js +72 -1
- package/src/mcp.js +384 -115
- package/src/refs.js +141 -0
- package/src/regions.js +203 -26
- package/src/screenmap.js +57 -16
- package/src/simctl.js +55 -2
- package/src/view.js +342 -0
|
@@ -0,0 +1,416 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Does the memory layer actually work, on a machine that is not the author's?
|
|
3
|
+
//
|
|
4
|
+
// The integration job asserts capture, input and OCR. It has never touched the
|
|
5
|
+
// layer above them — the screen map, element refs, the transition graph,
|
|
6
|
+
// verdicts, saved flows, goto — and that is precisely where every expensive bug
|
|
7
|
+
// in this project has lived: a state version that had drifted so every command
|
|
8
|
+
// respawned the daemon, `tap <label>` crashing on any screen the graph
|
|
9
|
+
// recognised, a tap and a type sharing one edge, a halted flow reporting
|
|
10
|
+
// success. Every one of those was found by hand or by somebody else running the
|
|
11
|
+
// tool. None was found by CI.
|
|
12
|
+
//
|
|
13
|
+
// Two rules this file follows, both learned from the step next to it in
|
|
14
|
+
// ci.yml, which was wrong three times:
|
|
15
|
+
//
|
|
16
|
+
// 1. Assert simframe's own bookkeeping, not iOS's behaviour. A test that
|
|
17
|
+
// needs Notification Center to open is measuring iOS.
|
|
18
|
+
// 2. Prefer actions whose effect is already proven. `openUrl` launches
|
|
19
|
+
// Safari, which every simulator has; `home` from inside an app always
|
|
20
|
+
// leaves it. Those two make a closed loop, so every pass starts from the
|
|
21
|
+
// same screen.
|
|
22
|
+
//
|
|
23
|
+
// idb is not installed on a hosted runner, so everything here runs OCR-only.
|
|
24
|
+
// That is deliberate: a warm flow makes zero accessibility reads (measured in
|
|
25
|
+
// docs/BENCHMARKS.md), and this is what holds that claim up.
|
|
26
|
+
import { execFile } from 'node:child_process';
|
|
27
|
+
import { promisify } from 'node:util';
|
|
28
|
+
import fs from 'node:fs';
|
|
29
|
+
import os from 'node:os';
|
|
30
|
+
import path from 'node:path';
|
|
31
|
+
import { fileURLToPath } from 'node:url';
|
|
32
|
+
|
|
33
|
+
const run = promisify(execFile);
|
|
34
|
+
const CLI = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'src', 'cli.js');
|
|
35
|
+
const device = process.argv.find((a) => a.startsWith('--device='))?.slice('--device='.length);
|
|
36
|
+
const FLOW_NAME = 'ci-memory-loop';
|
|
37
|
+
/** The graph converges over a few passes as variants are admitted; it is not instant. */
|
|
38
|
+
const CONVERGE_PASSES = 3;
|
|
39
|
+
/** Breathing room between passes. Each one launches Safari; a simulator asked to
|
|
40
|
+
* do that as fast as it can is being stress-tested, which is not the subject. */
|
|
41
|
+
const BETWEEN_PASSES_MS = 1200;
|
|
42
|
+
|
|
43
|
+
let failures = 0;
|
|
44
|
+
function check(ok, label, detail = '') {
|
|
45
|
+
if (!ok) failures += 1;
|
|
46
|
+
console.log(`${ok ? 'ok ' : 'FAIL'} ${label}${detail ? ` — ${detail}` : ''}`);
|
|
47
|
+
return ok;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* `expectFail` asserts a non-zero exit; `allowFail` tolerates one.
|
|
52
|
+
*
|
|
53
|
+
* The difference matters more than it looks. `simframe do` exits non-zero when
|
|
54
|
+
* a step does not verify, which is correct and is sometimes exactly the outcome
|
|
55
|
+
* under test — so the JSON has to be readable either way, or the test cannot
|
|
56
|
+
* tell "the flow reported a wrong turn" from "the command blew up".
|
|
57
|
+
*/
|
|
58
|
+
async function cli(args, { expectFail = false, allowFail = false } = {}) {
|
|
59
|
+
const full = device ? [...args, `--device=${device}`] : args;
|
|
60
|
+
try {
|
|
61
|
+
const { stdout } = await run('node', [CLI, ...full], { timeout: 180_000, maxBuffer: 32 << 20 });
|
|
62
|
+
if (expectFail) throw Object.assign(new Error('expected a non-zero exit'), { unexpectedSuccess: true, stdout });
|
|
63
|
+
return stdout;
|
|
64
|
+
} catch (err) {
|
|
65
|
+
if (err.unexpectedSuccess) throw err;
|
|
66
|
+
if (expectFail || allowFail) return `${err.stdout ?? ''}${err.stderr ?? ''}`;
|
|
67
|
+
const why = (err.stdout || err.stderr || err.message || '').trim();
|
|
68
|
+
throw new Error(`simframe ${full.join(' ')} failed: ${why.slice(0, 400)}`);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
async function json(args, opts) {
|
|
73
|
+
const out = await cli([...args, '--json'], opts);
|
|
74
|
+
try {
|
|
75
|
+
return JSON.parse(out);
|
|
76
|
+
} catch {
|
|
77
|
+
throw new Error(`simframe ${args.join(' ')} --json did not return JSON: ${out.trim().slice(0, 200)}`);
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* The simulator's display can briefly become unreadable — the daemon says so
|
|
83
|
+
* rather than serving a stale frame, which is the right behaviour and a
|
|
84
|
+
* transient condition. Retry those, and only those.
|
|
85
|
+
*/
|
|
86
|
+
const TRANSIENT = /did not produce a frame|display surface could not be read|no frames buffered/i;
|
|
87
|
+
|
|
88
|
+
async function jsonRetry(args, opts, attempts = 3) {
|
|
89
|
+
let last;
|
|
90
|
+
for (let i = 0; i < attempts; i += 1) {
|
|
91
|
+
try {
|
|
92
|
+
return await json(args, opts);
|
|
93
|
+
} catch (err) {
|
|
94
|
+
last = err;
|
|
95
|
+
if (!TRANSIENT.test(err.message)) throw err;
|
|
96
|
+
console.log(` (capture dropped out; retrying \`${args.join(' ')}\`)`);
|
|
97
|
+
await new Promise((r) => setTimeout(r, 1500));
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
throw last;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
const markHash = async () => (await jsonRetry(['mark'])).hash;
|
|
104
|
+
|
|
105
|
+
function writeFlow(name, steps) {
|
|
106
|
+
const file = path.join(os.tmpdir(), name);
|
|
107
|
+
fs.writeFileSync(file, JSON.stringify(steps));
|
|
108
|
+
return file;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
// A closed loop: openUrl puts Safari in front, home leaves it. Both ends are
|
|
112
|
+
// screens the graph can learn, and every pass starts where the last one ended.
|
|
113
|
+
const LOOP = writeFlow('simframe-ci-loop.json', [
|
|
114
|
+
{ openUrl: 'https://example.com' },
|
|
115
|
+
{ button: 'home' },
|
|
116
|
+
]);
|
|
117
|
+
// Leaving whatever screen the map was read on.
|
|
118
|
+
//
|
|
119
|
+
// Four versions of this proved nothing, each for the same reason: the harness
|
|
120
|
+
// guessed where the device was standing and guessed wrong.
|
|
121
|
+
//
|
|
122
|
+
// 1. `home` does not leave the home screen.
|
|
123
|
+
// 2. Opening Safari does not leave Safari.
|
|
124
|
+
// 3. Adding a Settings leaver does not leave Settings — it walks straight back
|
|
125
|
+
// to the screen the refs were numbered on, and the guard then *correctly*
|
|
126
|
+
// resolves the ref, which reads as the guard being broken.
|
|
127
|
+
// 4. Two pixel hashes are not evidence of anything when both are degenerate. A
|
|
128
|
+
// run here went `0000000000 -> 10ffffffff` — black, then uniform — and the
|
|
129
|
+
// precondition "the screen changed" passed on two hashes that cannot tell
|
|
130
|
+
// any screen from any other. This project has learned that lesson twice
|
|
131
|
+
// before, in the fingerprint and in the ref guard itself.
|
|
132
|
+
//
|
|
133
|
+
// So this no longer guesses. It puts the device on a named screen, reads the
|
|
134
|
+
// refs there, then puts it on a different named screen — two different apps, so
|
|
135
|
+
// they cannot be the same screen — and the pixel hash is used only as a
|
|
136
|
+
// corroborating signal, and only when it is informative.
|
|
137
|
+
//
|
|
138
|
+
// The pause is not padding. `settle` waits for a change and then for stillness,
|
|
139
|
+
// and called before the launch animation has begun it returns at once — so the
|
|
140
|
+
// map was read on a screen still arriving, the refs were numbered on that, and
|
|
141
|
+
// by the time the very next command ran simframe did not recognise where it
|
|
142
|
+
// was. The guard was right; the harness had numbered a ghost.
|
|
143
|
+
const AT_HOME_SCREEN = writeFlow('simframe-ci-at-reminders.json',
|
|
144
|
+
[{ launch: { value: 'com.apple.reminders', relaunch: true } }, { pause: 1600 }, { settle: true }]);
|
|
145
|
+
const AT_OTHER_SCREEN = writeFlow('simframe-ci-at-contacts.json',
|
|
146
|
+
[{ launch: { value: 'com.apple.MobileAddressBook', relaunch: true } }, { pause: 1600 }, { settle: true }]);
|
|
147
|
+
|
|
148
|
+
// A hash of one repeated character carries no information: an all-black screen
|
|
149
|
+
// and an all-white one are each a perfectly stable nothing, and two of them are
|
|
150
|
+
// within any tolerance of each other.
|
|
151
|
+
const informativeHash = (h) => typeof h === 'string' && h.length > 1 && !/^(.)\1*$/.test(h);
|
|
152
|
+
|
|
153
|
+
// Before anything else: is the device actually alive? A blank or wedged
|
|
154
|
+
// simulator produces "the display surface could not be read" on every call, and
|
|
155
|
+
// every check below then fails for a reason that has nothing to do with the
|
|
156
|
+
// memory layer. Diagnose it once, up front.
|
|
157
|
+
console.log('--- the device is alive ---');
|
|
158
|
+
const health = await jsonRetry(['state']);
|
|
159
|
+
check(/^[0-9a-f]{32}$/.test(health.hash ?? ''), 'capture is producing frames',
|
|
160
|
+
`frame #${health.seq}, ${health.width}x${health.height}`);
|
|
161
|
+
if (failures) {
|
|
162
|
+
console.error('\nthe device is not producing frames — nothing below would mean anything.');
|
|
163
|
+
console.error('a wedged or blank simulator needs restarting; `simframe doctor` says which layer is down.');
|
|
164
|
+
process.exit(1);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
console.log('\n--- the screen map ---');
|
|
168
|
+
await jsonRetry(['do', LOOP], { allowFail: true });
|
|
169
|
+
const map = await jsonRetry(['ui']);
|
|
170
|
+
|
|
171
|
+
// Report what actually answered rather than asserting the runner's situation.
|
|
172
|
+
// This line used to read "with no accessibility tree available" unconditionally,
|
|
173
|
+
// which is true on a hosted runner and a lie on a developer's machine.
|
|
174
|
+
const sources = map.sources ?? [];
|
|
175
|
+
// If this fails, the next question is always "which layer was missing, and
|
|
176
|
+
// why" — so answer it here rather than sending someone to the daemon log.
|
|
177
|
+
const degraded = map.degraded ?? [];
|
|
178
|
+
check(Array.isArray(map.elements) && map.elements.length > 0,
|
|
179
|
+
'the screen map has elements',
|
|
180
|
+
`${map.elements?.length ?? 0} element(s)${sources.length ? ` from ${sources.join('+')}` : ''}`
|
|
181
|
+
+ (degraded.length ? ` — degraded: ${degraded.join('; ')}` : ''));
|
|
182
|
+
check(/^[0-9a-f]{32}$/.test(map.screen?.hash ?? ''),
|
|
183
|
+
'the screen has a structural identity', map.screen?.hash?.slice(0, 12));
|
|
184
|
+
check(Number.isFinite(map.points?.width) && Number.isFinite(map.points?.height),
|
|
185
|
+
'the map knows the screen size in points', `${map.points?.width}x${map.points?.height}pt`);
|
|
186
|
+
|
|
187
|
+
const refs = (map.elements ?? []).map((e) => e.ref);
|
|
188
|
+
check(refs.every((r, i) => r === i + 1),
|
|
189
|
+
'refs are numbered 1..n with no gaps', `#1..#${refs.length}`);
|
|
190
|
+
check((map.elements ?? []).every((e) =>
|
|
191
|
+
Number.isInteger(e.x) && Number.isInteger(e.y)
|
|
192
|
+
&& e.y >= 0 && e.y <= map.points.height && e.x >= 0 && e.x <= map.points.width),
|
|
193
|
+
'every element has a tap point on the screen');
|
|
194
|
+
check(!(map.elements ?? []).some((e) => e.region === 'status-bar'),
|
|
195
|
+
'the status bar is not offered as something to tap');
|
|
196
|
+
|
|
197
|
+
console.log('\n--- element refs ---');
|
|
198
|
+
// Re-read on a screen we chose, rather than on whatever the device happened to
|
|
199
|
+
// be showing when this script started. `map` above is still the map of the
|
|
200
|
+
// as-found screen, and everything asserted about its shape holds either way.
|
|
201
|
+
await jsonRetry(['do', AT_HOME_SCREEN], { allowFail: true });
|
|
202
|
+
const refMap = await jsonRetry(['ui']);
|
|
203
|
+
const first = refMap.elements?.[0];
|
|
204
|
+
if (first) {
|
|
205
|
+
// A ref must resolve to exactly the point the map published, or the number in
|
|
206
|
+
// front of a row means nothing.
|
|
207
|
+
const found = await jsonRetry(['find', `#${first.ref}`]);
|
|
208
|
+
check(found.target?.x === first.x && found.target?.y === first.y,
|
|
209
|
+
`#${first.ref} resolves to the point the map gave it`,
|
|
210
|
+
`(${found.target?.x},${found.target?.y}) vs (${first.x},${first.y})`);
|
|
211
|
+
|
|
212
|
+
// Now leave that screen WITHOUT re-reading it: `--json` skips the end-state
|
|
213
|
+
// map, so the ref table still describes the screen we have left.
|
|
214
|
+
const before = await markHash();
|
|
215
|
+
await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
|
|
216
|
+
const after = await markHash();
|
|
217
|
+
// Two different apps are two different screens by construction. The pixel
|
|
218
|
+
// hashes only have to agree with that, and they only get a say when they are
|
|
219
|
+
// informative enough to have one.
|
|
220
|
+
const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
|
|
221
|
+
check(moved, 'the screen actually changed before testing the stale ref',
|
|
222
|
+
`${before.slice(0, 10)} -> ${after.slice(0, 10)}`
|
|
223
|
+
+ (informativeHash(before) && informativeHash(after) ? '' : ' (degenerate hash: not evidence either way)'));
|
|
224
|
+
// Only assert the guard if the precondition actually held. Running it anyway
|
|
225
|
+
// reports "the stale-ref guard failed" for a device that never left the
|
|
226
|
+
// screen, which is a false accusation against the one layer this file exists
|
|
227
|
+
// to defend — and it is how this check has failed twice.
|
|
228
|
+
if (moved) {
|
|
229
|
+
const stale = await cli(['find', `#${first.ref}`], { expectFail: true });
|
|
230
|
+
check(/different screen|read the screen/i.test(stale),
|
|
231
|
+
'a ref numbered on another screen refuses instead of tapping those coordinates',
|
|
232
|
+
stale.trim().split('\n')[0]?.slice(0, 90));
|
|
233
|
+
}
|
|
234
|
+
} else {
|
|
235
|
+
check(false, 'element refs', 'no elements to number');
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
console.log('\n--- the transition graph learns ---');
|
|
239
|
+
// What this asserts, and what it deliberately does not.
|
|
240
|
+
//
|
|
241
|
+
// It asserts that edges get recorded and that predictions start happening: the
|
|
242
|
+
// first pass reports `unverified` because nothing has been seen here before,
|
|
243
|
+
// and a later one reports `ok` because it has.
|
|
244
|
+
//
|
|
245
|
+
// It does NOT assert that every pass converges to all-`ok`. The springboard is
|
|
246
|
+
// the worst screen on the device to demand that of — it carries a live weather
|
|
247
|
+
// widget and a clock, so it has several genuine settled structures and a cap of
|
|
248
|
+
// four variants to hold them. Measured here: `home` read [ok, unverified],
|
|
249
|
+
// [ok, ok], [ok, unexpected-screen], [ok, unexpected-screen] over four passes.
|
|
250
|
+
// That is the known multiple-structures problem in docs/DEFERRED.md, not a
|
|
251
|
+
// regression, and a CI check that demanded otherwise would be flaky about
|
|
252
|
+
// something simframe does not claim.
|
|
253
|
+
//
|
|
254
|
+
// What IS worth pinning, and is exact, is that the run's own success flag
|
|
255
|
+
// agrees with its verdicts. `ok` used to mean no more than "nothing threw", so
|
|
256
|
+
// a flow halted at step 0 by a wrong turn reported "flow completed" with no
|
|
257
|
+
// error. That is a silent failure, and it is the one thing here that must never
|
|
258
|
+
// come back.
|
|
259
|
+
// A run that failed outright has no `results` at all — `--json` reports
|
|
260
|
+
// `{ok:false, error}`. Reaching into it crashed the harness with a TypeError
|
|
261
|
+
// pointing at this line, which says nothing about what went wrong on the
|
|
262
|
+
// device. A check script whose own failure mode is a stack trace is one more
|
|
263
|
+
// thing to debug at the moment you can least afford it.
|
|
264
|
+
const verdictsOf = (run) =>
|
|
265
|
+
Array.isArray(run?.results)
|
|
266
|
+
? run.results.map((r) => r.verification?.verdict ?? 'none')
|
|
267
|
+
: [`did not run: ${run?.error ?? 'no results'}`];
|
|
268
|
+
|
|
269
|
+
const passes = [];
|
|
270
|
+
for (let pass = 1; pass <= CONVERGE_PASSES; pass += 1) {
|
|
271
|
+
const run = await jsonRetry(['do', LOOP], { allowFail: true });
|
|
272
|
+
const verdicts = verdictsOf(run);
|
|
273
|
+
passes.push({ run, verdicts });
|
|
274
|
+
console.log(` pass ${pass}: ${run.ranSteps ?? 0}/${run.totalSteps ?? '?'} steps, verdicts [${verdicts.join(', ')}]`);
|
|
275
|
+
if (pass < CONVERGE_PASSES) await new Promise((r) => setTimeout(r, BETWEEN_PASSES_MS));
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
// An action the graph has never seen must say so. Asserting that from the loop
|
|
279
|
+
// above only works on a virgin graph, which a developer's machine is not after
|
|
280
|
+
// the first run — so ask with an action that is novel by construction: a URL
|
|
281
|
+
// nobody has opened before.
|
|
282
|
+
//
|
|
283
|
+
// Two things have to be true for this to mean anything, and getting either
|
|
284
|
+
// wrong makes the check lie rather than fail.
|
|
285
|
+
//
|
|
286
|
+
// The device must not already be on example.com. The URL is novel only in its
|
|
287
|
+
// query string, and that page renders identically whatever you put there, so
|
|
288
|
+
// from there the verdict is `no-visible-change` — the honest answer to what
|
|
289
|
+
// happened, and not the question being asked.
|
|
290
|
+
//
|
|
291
|
+
// And the positioning must happen in a *separate* run. Folding a launch into
|
|
292
|
+
// this flow made the check pass whenever the launch was unverified, whether or
|
|
293
|
+
// not the novel action was — a check that passes for the wrong reason is worse
|
|
294
|
+
// than one that fails, because nothing ever tells you.
|
|
295
|
+
await jsonRetry(['do', AT_HOME_SCREEN], { allowFail: true });
|
|
296
|
+
const NOVEL = writeFlow('simframe-ci-novel.json', [
|
|
297
|
+
{ openUrl: `https://example.com/?simframe-ci=${Date.now()}` },
|
|
298
|
+
]);
|
|
299
|
+
const novel = await jsonRetry(['do', NOVEL], { allowFail: true });
|
|
300
|
+
const novelSteps = Array.isArray(novel?.results) ? novel.results : [];
|
|
301
|
+
const novelVerdicts = novelSteps.length
|
|
302
|
+
? novelSteps.map((r) => r.verification?.verdict ?? (r.ok ? 'none' : `error: ${r.error}`))
|
|
303
|
+
: [`did not run: ${novel?.error ?? 'no results'}`];
|
|
304
|
+
// "Could this be tested" is a different question from "was the claim broken",
|
|
305
|
+
// and reporting the first as the second is how this file has spent the day
|
|
306
|
+
// accusing the layer underneath it. A device that crashed SpringBoard mid-run
|
|
307
|
+
// has not told us anything about the graph.
|
|
308
|
+
const novelRan = novelSteps.length > 0 && novelSteps.every((r) => r.ok !== false);
|
|
309
|
+
check(novelRan, 'the novel action ran at all', `[${novelVerdicts.join(', ')}]`);
|
|
310
|
+
if (novelRan) {
|
|
311
|
+
check(novelSteps.some((r) => r.verification?.verdict === 'unverified'),
|
|
312
|
+
'an action never taken here before is reported as unverified, not as verified',
|
|
313
|
+
`[${novelVerdicts.join(', ')}]`);
|
|
314
|
+
}
|
|
315
|
+
check(passes.some((p) => p.verdicts.includes('ok')),
|
|
316
|
+
'and once the graph has seen it, the outcome is predicted',
|
|
317
|
+
`pass ${passes.findIndex((p) => p.verdicts.includes('ok')) + 1}`);
|
|
318
|
+
|
|
319
|
+
// One direction only: a wrong turn must fail the run. The converse does not
|
|
320
|
+
// hold — a run can fail for reasons that are not wrong turns, such as a step
|
|
321
|
+
// that threw.
|
|
322
|
+
const silent = passes.filter((p) => p.verdicts.includes('unexpected-screen') && p.run.ok);
|
|
323
|
+
check(silent.length === 0,
|
|
324
|
+
'a run that reported a wrong turn never also reports success',
|
|
325
|
+
passes.map((p) => `${p.run.ok ? 'ok' : 'failed'}:[${p.verdicts.join(',')}]`).join(' '));
|
|
326
|
+
|
|
327
|
+
const halted = passes.find((p) => !p.run.ok);
|
|
328
|
+
if (halted) {
|
|
329
|
+
check(halted.run.ranSteps < halted.run.totalSteps
|
|
330
|
+
|| (Array.isArray(halted.run.results) && halted.run.results.some((r) => !r.ok)),
|
|
331
|
+
'and a halted run says which step stopped it',
|
|
332
|
+
`${halted.run.ranSteps}/${halted.run.totalSteps}`);
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
console.log('\n--- screens, and navigating to one ---');
|
|
336
|
+
const screens = await jsonRetry(['screens']);
|
|
337
|
+
check(Array.isArray(screens) && screens.length >= 2,
|
|
338
|
+
'more than one screen is remembered', `${screens.length ?? 0} known`);
|
|
339
|
+
|
|
340
|
+
// Worth writing down, because the first version of this check assumed
|
|
341
|
+
// otherwise: `screens` lists graph NODES, and a screen becomes a node when
|
|
342
|
+
// something is done *from* it. A destination that has only ever been arrived at
|
|
343
|
+
// is a hash on an edge, not yet a node — so the screen you are standing on
|
|
344
|
+
// after a flow is often not in this list, and that is correct.
|
|
345
|
+
const here = await jsonRetry(['ui']);
|
|
346
|
+
check(/^[0-9a-f]{32}$/.test(here.screen?.hash ?? ''),
|
|
347
|
+
'the screen we are standing on has an identity', here.screen?.hash?.slice(0, 8));
|
|
348
|
+
|
|
349
|
+
// The contract under test is refuse-rather-than-guess, and both halves of it
|
|
350
|
+
// are exact.
|
|
351
|
+
const nonsense = await jsonRetry(['goto', 'no-such-screen-zzz'], { allowFail: true });
|
|
352
|
+
check(nonsense.ok === false && nonsense.reason === 'unknown-screen',
|
|
353
|
+
'goto refuses a screen it has never seen', nonsense.reason);
|
|
354
|
+
check(Array.isArray(nonsense.known) && nonsense.known.length > 0,
|
|
355
|
+
'and says what it does know instead of guessing', `${nonsense.known?.length} screens offered`);
|
|
356
|
+
|
|
357
|
+
const target = screens[0];
|
|
358
|
+
const walked = await jsonRetry(['goto', target.hash], { allowFail: true });
|
|
359
|
+
const outcomes = ['no-route', 'unreplayable-edge', 'ambiguous', 'unknown-screen'];
|
|
360
|
+
check(walked.ok === true || outcomes.includes(walked.reason),
|
|
361
|
+
'and asked for a screen it knows, it either walks there or names why it cannot',
|
|
362
|
+
walked.ok ? (walked.already ? 'already there' : `walked ${walked.ranSteps} step(s)`) : walked.reason);
|
|
363
|
+
|
|
364
|
+
console.log('\n--- what may be saved, and what may not ---');
|
|
365
|
+
// A flow with an unverified step in it is a recording of something that may not
|
|
366
|
+
// have worked, and replaying it faithfully reproduces the doubt. So the
|
|
367
|
+
// contract runs both ways and both directions are checked against whatever the
|
|
368
|
+
// run actually produced, rather than assuming it verified.
|
|
369
|
+
const attempt = await jsonRetry(['do', LOOP, `--save=${FLOW_NAME}`], { allowFail: true });
|
|
370
|
+
const clean = attempt.results.every((r) => !r.verification || r.verification.verdict === 'ok');
|
|
371
|
+
if (clean) {
|
|
372
|
+
check(attempt.saved?.ok === true, 'a flow whose every step verified is saved',
|
|
373
|
+
`${attempt.saved?.steps} steps`);
|
|
374
|
+
} else {
|
|
375
|
+
check(attempt.saved?.ok === false && attempt.saved?.reason === 'unverified-steps',
|
|
376
|
+
'a flow with an unverified step is refused, not quietly saved',
|
|
377
|
+
`${attempt.saved?.reason} (${(attempt.saved?.verdicts ?? []).join(', ')})`);
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
// --force is the deliberate override, and it is what lets the replay machinery
|
|
381
|
+
// be tested without waiting on a screen that may never settle into one shape.
|
|
382
|
+
const forced = await jsonRetry(['do', LOOP, `--save=${FLOW_NAME}`, '--force'], { allowFail: true });
|
|
383
|
+
if (check(forced.saved?.ok === true, 'and --force saves it anyway', `${forced.saved?.steps} steps`)) {
|
|
384
|
+
const listed = await jsonRetry(['flow', 'list']);
|
|
385
|
+
check(listed.some((f) => f.name === FLOW_NAME), 'the saved flow is listed');
|
|
386
|
+
const replayed = await jsonRetry(['flow', 'run', FLOW_NAME], { allowFail: true });
|
|
387
|
+
check(replayed.ranSteps >= 1 && Array.isArray(replayed.results),
|
|
388
|
+
'and replays from disk with no model in the loop',
|
|
389
|
+
`${replayed.ranSteps}/${replayed.totalSteps} steps`);
|
|
390
|
+
const unknown = await cli(['flow', 'run', 'no-such-flow'], { expectFail: true });
|
|
391
|
+
check(/no flow/i.test(unknown), 'an unknown flow name is refused with what is known');
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
console.log('\n--- every command speaks JSON ---');
|
|
395
|
+
// The --json plumbing is per-command and hand-written, so one command quietly
|
|
396
|
+
// printing prose is exactly the kind of thing nothing else would catch.
|
|
397
|
+
for (const args of [['status'], ['state'], ['mark'], ['ui'], ['screens'], ['devices'], ['doctor'], ['flow', 'list'], ['recall']]) {
|
|
398
|
+
try {
|
|
399
|
+
const parsed = JSON.parse(await cli([...args, '--json']));
|
|
400
|
+
check(parsed !== null && parsed !== undefined, `simframe ${args.join(' ')} --json`);
|
|
401
|
+
} catch (err) {
|
|
402
|
+
check(false, `simframe ${args.join(' ')} --json`, err.message.slice(0, 120));
|
|
403
|
+
}
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
// Leave the device as it was found, minus the flow this test invented.
|
|
407
|
+
try {
|
|
408
|
+
const dir = path.join(process.env.SIMFRAME_HOME || path.join(os.homedir(), '.simframe'));
|
|
409
|
+
for (const udid of fs.readdirSync(dir)) {
|
|
410
|
+
const f = path.join(dir, udid, 'flows', `${FLOW_NAME}.json`);
|
|
411
|
+
if (fs.existsSync(f)) fs.rmSync(f);
|
|
412
|
+
}
|
|
413
|
+
} catch { /* nothing to clean up */ }
|
|
414
|
+
|
|
415
|
+
console.log(`\n${failures ? `${failures} check(s) failed` : 'every check passed'}`);
|
|
416
|
+
process.exit(failures ? 1 : 0);
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Do the two fingerprint distributions still separate?
|
|
3
|
+
//
|
|
4
|
+
// Screen identity is a Jaccard comparison of structural token sets against a
|
|
5
|
+
// threshold. That only works while two distributions stay apart: the similarity
|
|
6
|
+
// of a screen to *itself on a later visit*, and its similarity to *other
|
|
7
|
+
// screens*. Phase 6b measured them by hand at 0.41–1.00 against 0.00–0.31 — a
|
|
8
|
+
// gap of 0.11, with the threshold at 0.36 in the middle of it.
|
|
9
|
+
//
|
|
10
|
+
// By hand is not good enough for anything that changes what goes into a
|
|
11
|
+
// fingerprint. Region bands do: chrome labels are the only text in a token set,
|
|
12
|
+
// so how the bands are drawn decides which labels enter identity at all. This
|
|
13
|
+
// harness exists so that change can be measured either side instead of argued
|
|
14
|
+
// about.
|
|
15
|
+
//
|
|
16
|
+
// Every reading is taken COLD. `screenIdentity({fresh: true})` rebuilds the
|
|
17
|
+
// screen map rather than recalling it, because a recalled map returns the
|
|
18
|
+
// tokens of the visit that built it and would measure the cache rather than the
|
|
19
|
+
// fingerprint — which is exactly how 6b's warm row came out flatteringly wrong.
|
|
20
|
+
import fs from 'node:fs';
|
|
21
|
+
import * as actions from '../src/actions.js';
|
|
22
|
+
import * as fingerprint from '../src/fingerprint.js';
|
|
23
|
+
import * as graph from '../src/graph.js';
|
|
24
|
+
import * as api from '../src/index.js';
|
|
25
|
+
|
|
26
|
+
const arg = (name, fallback) => {
|
|
27
|
+
const hit = process.argv.find((a) => a.startsWith(`--${name}=`));
|
|
28
|
+
return hit ? hit.slice(name.length + 3) : fallback;
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
const tourFile = arg('tour');
|
|
32
|
+
const rounds = Number(arg('rounds', 3));
|
|
33
|
+
/**
|
|
34
|
+
* Above this, two consecutive tour screens are the same screen and the
|
|
35
|
+
* navigation between them failed. Deliberately well above the identity
|
|
36
|
+
* threshold: this is not "might be the same screen", it is "obviously is".
|
|
37
|
+
*/
|
|
38
|
+
const ARRIVAL_SUSPICION = 0.7;
|
|
39
|
+
const outFile = arg('out');
|
|
40
|
+
const device = arg('device');
|
|
41
|
+
const label = arg('label', 'unnamed');
|
|
42
|
+
|
|
43
|
+
if (!tourFile) {
|
|
44
|
+
console.error(`usage: node scripts/eval-fingerprint.mjs --tour=<tour.json> [--rounds=3] [--device=<udid>] [--out=<file>] [--label=<name>]
|
|
45
|
+
|
|
46
|
+
A tour is a JSON array of screens to visit in a cycle:
|
|
47
|
+
|
|
48
|
+
[{"name": "home", "steps": [{"button": "home"}]},
|
|
49
|
+
{"name": "browser", "steps": [{"openUrl": "https://example.com"}]}]
|
|
50
|
+
|
|
51
|
+
Each round walks the whole cycle, so every screen is left and re-arrived at
|
|
52
|
+
between readings — which is what makes "the same screen, revisited" a real
|
|
53
|
+
question rather than a re-read of the same frame.`);
|
|
54
|
+
process.exit(2);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const tour = JSON.parse(fs.readFileSync(tourFile, 'utf8'));
|
|
58
|
+
if (!Array.isArray(tour) || tour.length < 2) {
|
|
59
|
+
console.error('a tour needs at least two screens, or there are no different-screen pairs to measure');
|
|
60
|
+
process.exit(2);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const { device: dev } = await api.ensureDaemon(device);
|
|
64
|
+
console.log(`device: ${dev.name} (${dev.runtime})`);
|
|
65
|
+
console.log(`tour: ${tour.length} screens x ${rounds} rounds, every reading cold\n`);
|
|
66
|
+
|
|
67
|
+
/** @type {Array<{name: string, round: number, hash: string, tokens: string[], count: number}>} */
|
|
68
|
+
const readings = [];
|
|
69
|
+
/** Navigations that did not land before the reading was taken. */
|
|
70
|
+
const arrivalFailures = [];
|
|
71
|
+
|
|
72
|
+
for (let round = 1; round <= rounds; round += 1) {
|
|
73
|
+
for (const screen of tour) {
|
|
74
|
+
if (screen.steps?.length) {
|
|
75
|
+
// Verification is off: this measures fingerprints, and a wrong-turn
|
|
76
|
+
// verdict computed from the very tokens under test would be circular.
|
|
77
|
+
await actions.runScript(device, { steps: screen.steps, verify: false });
|
|
78
|
+
}
|
|
79
|
+
const id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
80
|
+
// Did we actually arrive? Two differently-named screens reading the same
|
|
81
|
+
// fingerprint means the navigation did not land before the reading was
|
|
82
|
+
// taken, and every distribution below it is then measuring the tour rather
|
|
83
|
+
// than the fingerprint. The first version of this harness did exactly that
|
|
84
|
+
// and reported that the distributions overlapped completely.
|
|
85
|
+
const previous = readings[readings.length - 1];
|
|
86
|
+
if (previous && previous.name !== screen.name) {
|
|
87
|
+
// Not just an identical hash. Two readings of the same screen can differ
|
|
88
|
+
// by a token and still obviously be the same screen — measured: a
|
|
89
|
+
// "springboard" reading that was actually Settings shared 11 of its 12
|
|
90
|
+
// tokens with the Settings reading beside it, and the harness passed it
|
|
91
|
+
// because the hashes differed. Anything this similar across a navigation
|
|
92
|
+
// means the navigation did not happen.
|
|
93
|
+
const s = fingerprint.similarity(previous.tokens, id.tokens ?? []);
|
|
94
|
+
if (s >= ARRIVAL_SUSPICION) {
|
|
95
|
+
arrivalFailures.push(
|
|
96
|
+
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
readings.push({
|
|
100
|
+
name: screen.name,
|
|
101
|
+
round,
|
|
102
|
+
hash: id.hash,
|
|
103
|
+
tokens: id.tokens ?? [],
|
|
104
|
+
count: (id.tokens ?? []).length,
|
|
105
|
+
settled: id.settled,
|
|
106
|
+
});
|
|
107
|
+
process.stdout.write(
|
|
108
|
+
` round ${round} ${screen.name.padEnd(14)} ${String(id.hash).slice(0, 10)} ${String((id.tokens ?? []).length).padStart(3)} tokens${id.settled ? '' : ' (never settled)'}\n`,
|
|
109
|
+
);
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
if (arrivalFailures.length) {
|
|
114
|
+
console.error(`\nFAIL ${arrivalFailures.length} reading(s) were taken on the previous screen:`);
|
|
115
|
+
for (const f of arrivalFailures) console.error(` ${f}`);
|
|
116
|
+
console.error('\nThe tour did not settle before being read, so the distributions below would');
|
|
117
|
+
console.error('measure the tour rather than the fingerprint. Fix the tour and re-run.');
|
|
118
|
+
console.error('A `settle` step defaults to mode "stable", which returns instantly in the');
|
|
119
|
+
console.error('moment before an animation begins — action steps already settle on their own.');
|
|
120
|
+
process.exit(1);
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const same = [];
|
|
124
|
+
const different = [];
|
|
125
|
+
for (let i = 0; i < readings.length; i += 1) {
|
|
126
|
+
for (let j = i + 1; j < readings.length; j += 1) {
|
|
127
|
+
const s = fingerprint.similarity(readings[i].tokens, readings[j].tokens);
|
|
128
|
+
(readings[i].name === readings[j].name ? same : different).push({
|
|
129
|
+
a: readings[i], b: readings[j], similarity: s,
|
|
130
|
+
});
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
const stats = (rows) => {
|
|
135
|
+
if (!rows.length) return null;
|
|
136
|
+
const v = rows.map((r) => r.similarity).sort((x, y) => x - y);
|
|
137
|
+
return {
|
|
138
|
+
n: v.length,
|
|
139
|
+
min: v[0],
|
|
140
|
+
max: v[v.length - 1],
|
|
141
|
+
median: v[v.length >> 1],
|
|
142
|
+
};
|
|
143
|
+
};
|
|
144
|
+
|
|
145
|
+
const s = stats(same);
|
|
146
|
+
const d = stats(different);
|
|
147
|
+
const gap = s && d ? s.min - d.max : null;
|
|
148
|
+
const threshold = graph.SIMILARITY_THRESHOLD;
|
|
149
|
+
|
|
150
|
+
const f = (x) => (x == null ? '—' : x.toFixed(2));
|
|
151
|
+
console.log(`\n${'distribution'.padEnd(26)} ${'n'.padStart(4)} ${'min'.padStart(6)} ${'median'.padStart(7)} ${'max'.padStart(6)}`);
|
|
152
|
+
console.log(`${'same screen, revisited'.padEnd(26)} ${String(s?.n ?? 0).padStart(4)} ${f(s?.min).padStart(6)} ${f(s?.median).padStart(7)} ${f(s?.max).padStart(6)}`);
|
|
153
|
+
console.log(`${'different screens'.padEnd(26)} ${String(d?.n ?? 0).padStart(4)} ${f(d?.min).padStart(6)} ${f(d?.median).padStart(7)} ${f(d?.max).padStart(6)}`);
|
|
154
|
+
console.log(`\ngap (same-min − different-max): ${f(gap)}`);
|
|
155
|
+
console.log(`threshold in use: ${threshold}`);
|
|
156
|
+
|
|
157
|
+
const separated = gap != null && gap > 0;
|
|
158
|
+
const thresholdInGap = s && d && threshold > d.max && threshold < s.min;
|
|
159
|
+
console.log(`${separated ? 'ok ' : 'FAIL'} the distributions ${separated ? 'separate' : 'OVERLAP — no threshold can tell these screens apart'}`);
|
|
160
|
+
console.log(`${thresholdInGap ? 'ok ' : 'WARN'} the threshold ${thresholdInGap ? 'sits inside the gap' : 'is NOT inside the gap'}`);
|
|
161
|
+
|
|
162
|
+
// The worst offenders, because a distribution is not actionable and a pair is.
|
|
163
|
+
const worstSame = [...same].sort((a, b) => a.similarity - b.similarity)[0];
|
|
164
|
+
const worstDifferent = [...different].sort((a, b) => b.similarity - a.similarity)[0];
|
|
165
|
+
if (worstSame) {
|
|
166
|
+
console.log(`\nweakest same-screen pair: ${worstSame.a.name} r${worstSame.a.round} vs r${worstSame.b.round} = ${f(worstSame.similarity)} (${worstSame.a.count} vs ${worstSame.b.count} tokens)`);
|
|
167
|
+
}
|
|
168
|
+
if (worstDifferent) {
|
|
169
|
+
console.log(`closest different-screen pair: ${worstDifferent.a.name} vs ${worstDifferent.b.name} = ${f(worstDifferent.similarity)}`);
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
// Chrome labels are the only text in a fingerprint, so which of them got in is
|
|
173
|
+
// the thing a band change actually moves.
|
|
174
|
+
const labels = new Set();
|
|
175
|
+
for (const r of readings) {
|
|
176
|
+
for (const t of r.tokens) {
|
|
177
|
+
if (t.includes('"')) labels.add(t.slice(t.indexOf('"') + 1, t.lastIndexOf('"')));
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
console.log(`\n${labels.size} distinct chrome label(s) entered identity: ${[...labels].sort().join(' · ') || '(none)'}`);
|
|
181
|
+
|
|
182
|
+
if (outFile) {
|
|
183
|
+
fs.writeFileSync(outFile, JSON.stringify({
|
|
184
|
+
label, device: dev.name, runtime: dev.runtime, rounds, at: Date.now(),
|
|
185
|
+
threshold, same: s, different: d, gap, separated, thresholdInGap,
|
|
186
|
+
labels: [...labels].sort(),
|
|
187
|
+
readings: readings.map(({ tokens, ...r }) => ({ ...r, tokenCount: tokens.length })),
|
|
188
|
+
}, null, 2));
|
|
189
|
+
console.log(`\nwrote ${outFile}`);
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
process.exit(separated ? 0 : 1);
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Keep server.json's version in step with package.json's.
|
|
3
|
+
//
|
|
4
|
+
// `npm version` only knows about package.json, and the MCP registry manifest
|
|
5
|
+
// carries the version twice — once at the top level and once inside the package
|
|
6
|
+
// entry. Every release therefore depended on remembering to hand-edit a second
|
|
7
|
+
// file between two commands, and the release that did not remember failed at
|
|
8
|
+
// the workflow's own agreement check.
|
|
9
|
+
//
|
|
10
|
+
// npm runs this as the `version` lifecycle script: after the bump, before the
|
|
11
|
+
// commit. It stages server.json so the version commit contains both files.
|
|
12
|
+
import { execFileSync } from 'node:child_process';
|
|
13
|
+
import fs from 'node:fs';
|
|
14
|
+
import path from 'node:path';
|
|
15
|
+
import { fileURLToPath } from 'node:url';
|
|
16
|
+
|
|
17
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
18
|
+
const pkg = JSON.parse(fs.readFileSync(path.join(ROOT, 'package.json'), 'utf8'));
|
|
19
|
+
const file = path.join(ROOT, 'server.json');
|
|
20
|
+
const server = JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
21
|
+
|
|
22
|
+
const before = { top: server.version, pkg: server.packages?.[0]?.version };
|
|
23
|
+
server.version = pkg.version;
|
|
24
|
+
if (!Array.isArray(server.packages) || !server.packages.length) {
|
|
25
|
+
console.error('server.json has no packages[] entry to version — has its shape changed?');
|
|
26
|
+
process.exit(1);
|
|
27
|
+
}
|
|
28
|
+
server.packages[0].version = pkg.version;
|
|
29
|
+
|
|
30
|
+
fs.writeFileSync(file, `${JSON.stringify(server, null, 2)}\n`);
|
|
31
|
+
console.log(`server.json ${before.top} / ${before.pkg} -> ${pkg.version} / ${pkg.version}`);
|
|
32
|
+
|
|
33
|
+
// Stage it, so `npm version` commits both files together. Harmless when run
|
|
34
|
+
// with --no-git-tag-version; the file is still correct either way.
|
|
35
|
+
try {
|
|
36
|
+
execFileSync('git', ['add', '--', file], { cwd: ROOT, stdio: 'pipe' });
|
|
37
|
+
} catch {
|
|
38
|
+
console.log('(could not stage server.json — commit it yourself)');
|
|
39
|
+
}
|