simframe 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +181 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +33 -3
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +216 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/scripts/probe-network.mjs +118 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +257 -7
- package/src/actions.js +1791 -44
- package/src/cli.js +216 -14
- package/src/control.js +1 -0
- package/src/fingerprint.js +43 -1
- package/src/graph.js +136 -7
- package/src/index.js +252 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +161 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +333 -32
- package/src/metrics.js +148 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +25 -1
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +25 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +215 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +161 -0
- package/src/view.js +375 -11
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/cli.js
CHANGED
|
@@ -3,7 +3,7 @@ import fs from 'node:fs';
|
|
|
3
3
|
import os from 'node:os';
|
|
4
4
|
import path from 'node:path';
|
|
5
5
|
import { runDaemon, DEFAULTS } from './daemon.js';
|
|
6
|
-
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
|
|
6
|
+
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, restartDevice, screenshot, toolchainChecks } from './platform/index.js';
|
|
7
7
|
import * as actions from './actions.js';
|
|
8
8
|
import * as analyze from './analyze.js';
|
|
9
9
|
import * as api from './index.js';
|
|
@@ -47,15 +47,23 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
47
47
|
simframe baseline list recorded runs per flow, and what is committed
|
|
48
48
|
simframe hpi [device] Human Parity Index, per flow and overall
|
|
49
49
|
simframe escalations [device] why simframe handed decisions back, by reason
|
|
50
|
+
simframe supervisions [device] local supervisor rulings, and what came of each
|
|
51
|
+
simframe revive [device] power-cycle a wedged device: stop, shutdown, boot, start, reset input
|
|
52
|
+
(--session=<id> narrows to one agent; the
|
|
53
|
+
ids are listed in the output. SIMFRAME_SESSION
|
|
54
|
+
names one, but only at process start — an
|
|
55
|
+
already-running MCP server cannot pick it up)
|
|
50
56
|
simframe devices list simulators
|
|
51
57
|
simframe doctor check that this machine can capture
|
|
52
58
|
(--strict, or SIMFRAME_STRICT=1, makes any
|
|
53
59
|
degraded layer a non-zero exit)
|
|
54
60
|
|
|
55
|
-
Selectors — anywhere a control is named
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
61
|
+
Selectors — anywhere a control is named, best first
|
|
62
|
+
"Save" a label or a phrase, resolved by intent (verbs, typos, synonyms,
|
|
63
|
+
icon-only controls by their common name). Start here.
|
|
64
|
+
#3 the number \`simframe ui\` gave it. Exact, but only inside the
|
|
65
|
+
round trip that numbered it — the screen moves and it does not.
|
|
66
|
+
@120,400 raw point coordinates. Last resort: it cannot tell you it missed.
|
|
59
67
|
|
|
60
68
|
Measuring against a human — the Human Parity Index
|
|
61
69
|
|
|
@@ -162,6 +170,25 @@ const num = (v, fallback) => (v == null ? fallback : Number(v));
|
|
|
162
170
|
* script that has to parse one command's prose and another's JSON will parse
|
|
163
171
|
* the prose wrong exactly once and then be trusted anyway.
|
|
164
172
|
*/
|
|
173
|
+
/**
|
|
174
|
+
* The machine-readable half of a failure.
|
|
175
|
+
*
|
|
176
|
+
* A refusal recognisable only by reading its prose is a refusal nobody can
|
|
177
|
+
* depend on. Our own CI asserted the stale-ref guard by matching three
|
|
178
|
+
* phrasings and went red when a fourth arrived — a *better* one, naming the
|
|
179
|
+
* label the number stood for. The sentence is for a person; these fields are
|
|
180
|
+
* the contract, and they live in one function because `find` reports its own
|
|
181
|
+
* failures and the top-level handler reports the rest.
|
|
182
|
+
*/
|
|
183
|
+
function failureJson(err) {
|
|
184
|
+
return {
|
|
185
|
+
ok: false,
|
|
186
|
+
error: err.message,
|
|
187
|
+
reason: metrics.escalationOf(err)?.reason ?? null,
|
|
188
|
+
...(err.staleRef ? { staleRef: true, staleKind: err.staleKind ?? null, staleLabel: err.staleLabel ?? null } : {}),
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
|
|
165
192
|
function emit(flags, json, lines) {
|
|
166
193
|
if (flags.json) {
|
|
167
194
|
console.log(JSON.stringify(json, null, 2));
|
|
@@ -171,11 +198,26 @@ function emit(flags, json, lines) {
|
|
|
171
198
|
if (body != null) console.log(Array.isArray(body) ? body.filter((l) => l != null).join('\n') : body);
|
|
172
199
|
}
|
|
173
200
|
|
|
174
|
-
/**
|
|
175
|
-
|
|
201
|
+
/**
|
|
202
|
+
* The end-state screen map, re-read rather than recalled, with its hint.
|
|
203
|
+
*
|
|
204
|
+
* Both halves were reported against Phase 11.5 and both were right. The map was
|
|
205
|
+
* rendered from whatever reading the flow already had, which is memory-first —
|
|
206
|
+
* so a trailing map could describe the screen as it was seconds ago, and the
|
|
207
|
+
* remedy in practice was a `ui --refresh` after nearly every call, which is a
|
|
208
|
+
* whole extra turn to save a few hundred milliseconds. Wrong way round.
|
|
209
|
+
*
|
|
210
|
+
* And the hint was only ever printed by the MCP server, so no CLI user could
|
|
211
|
+
* see it and no CLI run could test it.
|
|
212
|
+
*/
|
|
213
|
+
async function mapText(device, options, identity, { flowOk = true, escalated = false, refresh = true } = {}) {
|
|
176
214
|
try {
|
|
177
|
-
const m = await view.screenMap(device, {
|
|
178
|
-
|
|
215
|
+
const m = await view.screenMap(device, {
|
|
216
|
+
options,
|
|
217
|
+
refresh,
|
|
218
|
+
identity: refresh ? undefined : (identity?.entry ? identity : undefined),
|
|
219
|
+
});
|
|
220
|
+
return `${m.text}\n${view.hintFor(m, { flowOk, escalated })}`;
|
|
179
221
|
} catch (err) {
|
|
180
222
|
return `(could not read the screen: ${err.message})`;
|
|
181
223
|
}
|
|
@@ -231,6 +273,11 @@ async function main() {
|
|
|
231
273
|
if (flags.maxDim) options.maxDim = num(flags.maxDim);
|
|
232
274
|
if (flags.ringSize) options.ringSize = num(flags.ringSize);
|
|
233
275
|
if (flags.engine) options.engine = String(flags.engine);
|
|
276
|
+
// Per-command overrides for the two experiment knobs, so an A/B is an
|
|
277
|
+
// argument rather than a restart. `--sensor=ax-first`, `--planner=apple`.
|
|
278
|
+
if (flags.sensor) options.sensor = String(flags.sensor);
|
|
279
|
+
if (flags.planner) options.planner = String(flags.planner);
|
|
280
|
+
if (flags.supervisor) options.supervisor = String(flags.supervisor);
|
|
234
281
|
|
|
235
282
|
switch (command) {
|
|
236
283
|
case undefined:
|
|
@@ -335,6 +382,51 @@ async function main() {
|
|
|
335
382
|
return;
|
|
336
383
|
}
|
|
337
384
|
|
|
385
|
+
// The power cycle, when the narrow remedies are spent.
|
|
386
|
+
//
|
|
387
|
+
// Deliberately a command and not a behaviour. The capture loop tries two
|
|
388
|
+
// things — re-resolve the display port, then rebind the device — and then
|
|
389
|
+
// reports `stalled` and stops, because a capture loop that rebooted the
|
|
390
|
+
// device it was watching would be a tool reaching for the mains when a
|
|
391
|
+
// reading looks wrong. Restarting is the operator's call.
|
|
392
|
+
//
|
|
393
|
+
// But it was the operator's call *and* their four commands, remembered from
|
|
394
|
+
// a handoff note: stop the daemon, shut the device down, boot it and wait,
|
|
395
|
+
// start capture, rebuild the HID session. Done by hand three times in one
|
|
396
|
+
// afternoon, in that order, because any other order leaves a daemon holding
|
|
397
|
+
// a dead device. So the tool knows the order now; the decision is still
|
|
398
|
+
// yours.
|
|
399
|
+
case 'revive': {
|
|
400
|
+
const dev = await resolveDevice(device);
|
|
401
|
+
const say = (line) => { if (!flags.json) console.log(line); };
|
|
402
|
+
const steps = [];
|
|
403
|
+
const did = async (what, fn) => {
|
|
404
|
+
try { await fn(); steps.push({ step: what, ok: true }); say(` ok ${what}`); } catch (err) {
|
|
405
|
+
steps.push({ step: what, ok: false, error: err.message });
|
|
406
|
+
say(` .. ${what} — ${err.message.split('\n')[0]}`);
|
|
407
|
+
}
|
|
408
|
+
};
|
|
409
|
+
say(`reviving ${dev.name}`);
|
|
410
|
+
// Forced: the point of this command is that the device is wedged, so
|
|
411
|
+
// something is certainly still holding it.
|
|
412
|
+
await did('stopped the daemon', async () => { api.stopDaemon(dev.udid, { force: true }); });
|
|
413
|
+
// Through the boundary, which is the whole point of the boundary: the
|
|
414
|
+
// first version of this shelled out to `xcrun` from here and the test
|
|
415
|
+
// that forbids it failed immediately, correctly.
|
|
416
|
+
await did('restarted the device, and waited for the boot to finish',
|
|
417
|
+
() => restartDevice(dev.udid));
|
|
418
|
+
await did('started capture', () => api.ensureDaemon(dev.udid));
|
|
419
|
+
await did('rebuilt the HID session', () => input.resetSession(dev.udid));
|
|
420
|
+
const health = await api.getState(dev.udid).then((s) => s?.state ?? null).catch(() => null);
|
|
421
|
+
const alive = Boolean(health?.hash);
|
|
422
|
+
emit(flags, { ok: alive, device: dev.udid, steps }, alive
|
|
423
|
+
? `\n${dev.name} is producing frames again`
|
|
424
|
+
: `\n${dev.name} is still not producing frames. This is past what simframe can do —`
|
|
425
|
+
+ ' check Simulator.app is not showing an error, and see docs/DEFERRED.md item 95.');
|
|
426
|
+
if (!alive) process.exitCode = 1;
|
|
427
|
+
return;
|
|
428
|
+
}
|
|
429
|
+
|
|
338
430
|
case 'status': {
|
|
339
431
|
const udids = device
|
|
340
432
|
? [(await resolveDevice(device)).udid]
|
|
@@ -533,6 +625,7 @@ async function main() {
|
|
|
533
625
|
all: Boolean(flags.all),
|
|
534
626
|
refresh: Boolean(flags.refresh),
|
|
535
627
|
});
|
|
628
|
+
m.text = `${m.text}\n${view.hintFor(m)}`;
|
|
536
629
|
emit(
|
|
537
630
|
flags,
|
|
538
631
|
{
|
|
@@ -585,6 +678,7 @@ async function main() {
|
|
|
585
678
|
stableMs: num(flags.stableMs, 500),
|
|
586
679
|
timeoutMs: num(flags.timeoutMs, 8000),
|
|
587
680
|
continueOnError: Boolean(flags.continueOnError),
|
|
681
|
+
supervise: flags.supervise ? String(flags.supervise) : undefined,
|
|
588
682
|
options,
|
|
589
683
|
});
|
|
590
684
|
const saved = flags.save
|
|
@@ -592,7 +686,16 @@ async function main() {
|
|
|
592
686
|
: null;
|
|
593
687
|
// `--map=false` arrives as the string "false"; `--no-map` as true.
|
|
594
688
|
const wantMap = !flags.json && flags.noMap !== true && String(flags.map ?? 'true') !== 'false';
|
|
595
|
-
|
|
689
|
+
// Every local ruling, so a wrong one is correctable rather than
|
|
690
|
+
// mysterious — and the model's stated reason is shown as its claim.
|
|
691
|
+
for (const s_ of res.supervisions ?? []) {
|
|
692
|
+
console.log(`supervisor at step ${s_.index}: ${s_.decision} — ${s_.outcome}`
|
|
693
|
+
+ (s_.reason ? ` (it said: "${s_.reason}")` : ''));
|
|
694
|
+
}
|
|
695
|
+
const escalated = (res.results ?? []).some((r) => metrics.ESCALATING_VERDICTS.has(r.verification?.verdict));
|
|
696
|
+
const map = wantMap
|
|
697
|
+
? await mapText(flags.device, options, res.endScreen, { flowOk: res.ok, escalated })
|
|
698
|
+
: null;
|
|
596
699
|
emit(
|
|
597
700
|
flags,
|
|
598
701
|
{
|
|
@@ -812,7 +915,7 @@ async function main() {
|
|
|
812
915
|
],
|
|
813
916
|
);
|
|
814
917
|
} catch (err) {
|
|
815
|
-
emit(flags,
|
|
918
|
+
emit(flags, failureJson(err), err.message);
|
|
816
919
|
process.exitCode = 1;
|
|
817
920
|
}
|
|
818
921
|
return;
|
|
@@ -973,6 +1076,37 @@ async function main() {
|
|
|
973
1076
|
return;
|
|
974
1077
|
}
|
|
975
1078
|
|
|
1079
|
+
case 'supervisions': {
|
|
1080
|
+
const dev = await resolveDevice(flags.device);
|
|
1081
|
+
const records = metrics.readSupervisions(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
|
|
1082
|
+
const b = metrics.supervisionBreakdown(records);
|
|
1083
|
+
if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify({ ...b, records }, null, 2)}\n`);
|
|
1084
|
+
emit(flags, { ...b, records: flags.verbose ? records : undefined }, [
|
|
1085
|
+
`${b.total} supervisor ruling${b.total === 1 ? '' : 's'} on ${dev.name}`,
|
|
1086
|
+
b.total ? '' : 'Nothing has been judged on this device yet. The supervisor is off unless'
|
|
1087
|
+
+ ' SIMFRAME_SUPERVISOR=apple, and a ruling is only recorded when a step actually fails.',
|
|
1088
|
+
...Object.entries(b.decision_to_outcome)
|
|
1089
|
+
.sort((a, c) => c[1] - a[1])
|
|
1090
|
+
.map(([k, n]) => ` ${k.padEnd(28)} ${String(n).padStart(4)}`),
|
|
1091
|
+
b.total ? '' : null,
|
|
1092
|
+
b.total ? `sourced: ${Object.entries(b.by_from).map(([k, n]) => `${k} ${n}`).join(', ')}` : null,
|
|
1093
|
+
b.median_latency_ms != null ? `median latency: ${b.median_latency_ms}ms` : null,
|
|
1094
|
+
// Said out loud, because the first version of item 101 claimed its
|
|
1095
|
+
// measurement ran "on logs we already have" when nothing persisted a
|
|
1096
|
+
// ruling at all. This line is what stops that claim being made twice.
|
|
1097
|
+
b.total
|
|
1098
|
+
? `edges the graph had timed: ${b.p95_known}/${b.total}`
|
|
1099
|
+
+ (b.p95_unknown
|
|
1100
|
+
? ` — ${b.p95_unknown} ruling(s) are on edges with no p95, so they cannot take part in 101's comparison`
|
|
1101
|
+
: '')
|
|
1102
|
+
: null,
|
|
1103
|
+
b.sessions.length > 1
|
|
1104
|
+
? `WARNING ${b.sessions.length} sessions are pooled here; two agents on one device write one file`
|
|
1105
|
+
: null,
|
|
1106
|
+
].filter((l) => l !== null).join('\n'));
|
|
1107
|
+
break;
|
|
1108
|
+
}
|
|
1109
|
+
|
|
976
1110
|
case 'escalations': {
|
|
977
1111
|
const dev = await resolveDevice(flags.device);
|
|
978
1112
|
const records = metrics.readEscalations(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
|
|
@@ -1047,6 +1181,10 @@ async function main() {
|
|
|
1047
1181
|
json: Boolean(flags.json),
|
|
1048
1182
|
strict: Boolean(flags.strict) || process.env.SIMFRAME_STRICT === '1',
|
|
1049
1183
|
device: flags.device,
|
|
1184
|
+
// So `doctor --sensor=ax-first --planner=apple` reports the mode the
|
|
1185
|
+
// caller is about to use, not the one the environment happens to hold.
|
|
1186
|
+
// Confirming the mode before a run is the whole reason to read this.
|
|
1187
|
+
options,
|
|
1050
1188
|
});
|
|
1051
1189
|
return;
|
|
1052
1190
|
}
|
|
@@ -1123,7 +1261,7 @@ async function blackScreenProbe(udid) {
|
|
|
1123
1261
|
}
|
|
1124
1262
|
}
|
|
1125
1263
|
|
|
1126
|
-
async function doctor({ json = false, strict = false, device } = {}) {
|
|
1264
|
+
async function doctor({ json = false, strict = false, device, options = {} } = {}) {
|
|
1127
1265
|
const checks = [];
|
|
1128
1266
|
// `level` is 'ok' | 'warn' | 'fail'. A warn means it works but not the way it
|
|
1129
1267
|
// should — the exact state that used to be invisible.
|
|
@@ -1172,6 +1310,55 @@ async function doctor({ json = false, strict = false, device } = {}) {
|
|
|
1172
1310
|
add('on-device OCR', 'warn', err.message, { key: 'ocr.available', value: false });
|
|
1173
1311
|
}
|
|
1174
1312
|
|
|
1313
|
+
// Which sensors a read asks for, and what the recognition-level flag can and
|
|
1314
|
+
// cannot reach. Stated because `SIMFRAME_OCR` looks like it configures the
|
|
1315
|
+
// OCR everyone uses and does not: the daemon reads text in-process off the
|
|
1316
|
+
// framebuffer and owns its own recognition level, so the flag only reaches
|
|
1317
|
+
// the no-daemon fallback in `native/ocr.swift`.
|
|
1318
|
+
try {
|
|
1319
|
+
const api = await import('./index.js');
|
|
1320
|
+
const ocrMod = await import('./ocr.js');
|
|
1321
|
+
const mode = api.sensorMode(options);
|
|
1322
|
+
add('sensor mode', 'ok',
|
|
1323
|
+
mode === 'ax-first'
|
|
1324
|
+
? 'ax-first — the tree alone (~50ms), paying for OCR only when a resolve fails'
|
|
1325
|
+
: 'full — accessibility and OCR fused on every read (~164ms)',
|
|
1326
|
+
{ key: 'sensor.mode', value: mode });
|
|
1327
|
+
add('OCR level', 'ok', `${ocrMod.level()} (fallback helper only; the daemon owns its own)`, {
|
|
1328
|
+
key: 'ocr.level',
|
|
1329
|
+
value: ocrMod.level(),
|
|
1330
|
+
});
|
|
1331
|
+
} catch { /* reported by the layers above */ }
|
|
1332
|
+
|
|
1333
|
+
// The local supervisor. Behind the hands and in front of the reasoner, and
|
|
1334
|
+
// able to say only wait/retry/stop.
|
|
1335
|
+
try {
|
|
1336
|
+
const supervisor = await import('./supervisor.js');
|
|
1337
|
+
const st = await supervisor.status(options);
|
|
1338
|
+
add('local supervisor', 'ok', `${st.supervisor} — ${st.detail}`, {
|
|
1339
|
+
key: 'supervisor.backend',
|
|
1340
|
+
value: st.supervisor,
|
|
1341
|
+
});
|
|
1342
|
+
supervisor.close();
|
|
1343
|
+
} catch (err) {
|
|
1344
|
+
add('local supervisor', 'ok', `none — ${err.message}`, { key: 'supervisor.backend', value: 'none' });
|
|
1345
|
+
}
|
|
1346
|
+
|
|
1347
|
+
// The local planner tier. `none` is the normal answer and not a fault: it is
|
|
1348
|
+
// off unless SIMFRAME_PLANNER asks for it, and it only ever reorders
|
|
1349
|
+
// candidates that exploration was going to try anyway.
|
|
1350
|
+
try {
|
|
1351
|
+
const planner = await import('./planner.js');
|
|
1352
|
+
const st = await planner.status(options);
|
|
1353
|
+
add('local planner', 'ok', `${st.planner} — ${st.detail}`, {
|
|
1354
|
+
key: 'planner.backend',
|
|
1355
|
+
value: st.planner,
|
|
1356
|
+
});
|
|
1357
|
+
planner.close();
|
|
1358
|
+
} catch (err) {
|
|
1359
|
+
add('local planner', 'ok', `none — ${err.message}`, { key: 'planner.backend', value: 'none' });
|
|
1360
|
+
}
|
|
1361
|
+
|
|
1175
1362
|
try {
|
|
1176
1363
|
let booted = await bootedDevices();
|
|
1177
1364
|
// Respect --device. Without this, doctor reports on every booted simulator,
|
|
@@ -1398,13 +1585,28 @@ async function doctor({ json = false, strict = false, device } = {}) {
|
|
|
1398
1585
|
process.exitCode = failed.length || (strict && warned.length) ? 1 : 0;
|
|
1399
1586
|
}
|
|
1400
1587
|
|
|
1401
|
-
|
|
1588
|
+
// A long-lived local helper must not decide when the CLI exits. It is closed
|
|
1589
|
+
// after every command, whether or not one was ever started — `close()` on an
|
|
1590
|
+
// unopened planner is a no-op, and leaving it open made a finished flow hang.
|
|
1591
|
+
const closeHelpers = async () => {
|
|
1592
|
+
try {
|
|
1593
|
+
const planner = await import('./planner.js');
|
|
1594
|
+
planner.close();
|
|
1595
|
+
const supervisor = await import('./supervisor.js');
|
|
1596
|
+
supervisor.close();
|
|
1597
|
+
} catch { /* nothing to close */ }
|
|
1598
|
+
};
|
|
1599
|
+
|
|
1600
|
+
main().then(closeHelpers, async (err) => {
|
|
1601
|
+
await closeHelpers();
|
|
1602
|
+
throw err;
|
|
1603
|
+
}).catch((err) => {
|
|
1402
1604
|
// A caller that asked for JSON gets JSON, failures included. Printing prose
|
|
1403
1605
|
// here handed `JSON.parse` a SyntaxError instead of a reason, so a script
|
|
1404
1606
|
// could not tell "the daemon lost the display" from "simframe is broken" —
|
|
1405
1607
|
// which is the whole point of a machine-readable interface.
|
|
1406
1608
|
if (process.argv.includes('--json')) {
|
|
1407
|
-
process.stdout.write(`${JSON.stringify(
|
|
1609
|
+
process.stdout.write(`${JSON.stringify(failureJson(err), null, 2)}\n`);
|
|
1408
1610
|
} else {
|
|
1409
1611
|
process.stderr.write(`simframe: ${err.message}\n`);
|
|
1410
1612
|
}
|
package/src/control.js
CHANGED
|
@@ -65,6 +65,7 @@ export const swipe = (udid, from, to, opts = {}) =>
|
|
|
65
65
|
export const type = (udid, text) => request(udid, { action: 'type', text });
|
|
66
66
|
export const paste = (udid, text) => request(udid, { action: 'paste', text });
|
|
67
67
|
export const press = (udid, button) => request(udid, { action: 'press', button });
|
|
68
|
+
export const key = (udid, usage, modifiers = []) => request(udid, { action: 'key', usage, modifiers });
|
|
68
69
|
export const status = (udid) => request(udid, { action: 'status' });
|
|
69
70
|
export const resetInput = (udid) => request(udid, { action: 'resetInput' });
|
|
70
71
|
export const longPress = (udid, x, y, opts = {}) => request(udid, { action: 'longPress', x, y, ...opts });
|
package/src/fingerprint.js
CHANGED
|
@@ -30,8 +30,50 @@ import * as regions from './regions.js';
|
|
|
30
30
|
* list screen. Same rules, different input, therefore different hashes —
|
|
31
31
|
* and a stored hash that can never match again is the quietest kind of
|
|
32
32
|
* wrong, which is what this counter exists to prevent.
|
|
33
|
+
* 5 — a phantom keyboard was deleting screens' content from their identity. A
|
|
34
|
+
* dozen short text rows of uniform height stacked low on a read-only
|
|
35
|
+
* summary satisfied every size-and-uniformity test for a keyboard, and
|
|
36
|
+
* `tokens` discards everything below `keyboardTop` — so two screens of one
|
|
37
|
+
* wizard, sharing a nav title and a step indicator, collapsed onto a single
|
|
38
|
+
* hash. `detectKeyboardTop` now requires the small uniform boxes to be
|
|
39
|
+
* key-shaped. Every screen with content in its lower half hashes
|
|
40
|
+
* differently, so the stored graph and maps must go.
|
|
41
|
+
* 6 — the opposite half of the same bug, and it took a recorded screen to see.
|
|
42
|
+
* `KEYBOARD_MIN_FRACTION` is a *detection window*, not a keyboard's height,
|
|
43
|
+
* and its edge was being used as the boundary — so on an iPhone 17 Pro with
|
|
44
|
+
* the software keyboard up, the window starts at y=629 while the `q`–`p`
|
|
45
|
+
* row's frame top is **590**, and that whole row fell outside it. Ten
|
|
46
|
+
* keyboard keys were reported as page content and counted into the screen's
|
|
47
|
+
* identity, in the same map that said `keyboard up`. The boundary now
|
|
48
|
+
* extends upward while the rows above keep being key-shaped, which page
|
|
49
|
+
* content is not. Any screen fingerprinted with a keyboard up hashes
|
|
50
|
+
* differently, so the stored graph and maps must go again.
|
|
51
|
+
*
|
|
52
|
+
* 7 — a screen whose own name is an iOS **large title** had no name at all in
|
|
53
|
+
* its identity. Chrome labels are the only text these tokens keep, and a
|
|
54
|
+
* large title is drawn tight against the content it heads — 79 pt of inset
|
|
55
|
+
* above it and 5.3 pt below, against a boundary bar of 66.5 — so the top
|
|
56
|
+
* chrome detector, which looks for the gap *beneath* a bar, never found it
|
|
57
|
+
* and the title was discarded as content. Measured on the Settings root in
|
|
58
|
+
* both sensor modes: **0 named tokens**, and the same for Contacts and
|
|
59
|
+
* Reminders. On a hosted runner two sparse nameless readings then matched
|
|
60
|
+
* exactly and one hash stood for two different screens. `regions.bands`
|
|
61
|
+
* now finds a large title by the inset above it; every affected screen
|
|
62
|
+
* hashes differently, so stored graphs and maps go again.
|
|
63
|
+
*
|
|
64
|
+
* 8 — the same rule, keyed on the wrong thing. Its "is this a real inset"
|
|
65
|
+
* test compared the gap above the title against the screen's **median row
|
|
66
|
+
* gap**, so whether a system-drawn title counted as chrome depended on how
|
|
67
|
+
* many rows happened to sit below it. Caught on the React Native testbed's
|
|
68
|
+
* first day, with two screens of one app: a list of 24 rows has a median
|
|
69
|
+
* gap of 0 and the rule fired, a list of 4 above a tab bar has a median gap
|
|
70
|
+
* of **414** and it did not. Same title, same 62.9pt inset, opposite
|
|
71
|
+
* answers — so one screen carried a name and the other did not, and the
|
|
72
|
+
* graph merged them. Both bounds are absolute now. Screens that were
|
|
73
|
+
* missed at 7 hash differently at 8, and unlike a stale hash that matches
|
|
74
|
+
* nothing, these matched the *wrong* thing.
|
|
33
75
|
*/
|
|
34
|
-
export const TOKEN_RULES_VERSION =
|
|
76
|
+
export const TOKEN_RULES_VERSION = 8;
|
|
35
77
|
|
|
36
78
|
/** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
|
|
37
79
|
export const GRID = 24;
|
package/src/graph.js
CHANGED
|
@@ -258,6 +258,28 @@ export function nearestScreen(udid, screen, { threshold = SIMILARITY_THRESHOLD }
|
|
|
258
258
|
for (const node of nodes) {
|
|
259
259
|
for (const f of fingerprintsOf(node)) {
|
|
260
260
|
if (!f.tokens?.length) continue;
|
|
261
|
+
// A screen that says it is called something else is not this screen.
|
|
262
|
+
//
|
|
263
|
+
// Similarity weighs every token equally, and a chrome label is not an
|
|
264
|
+
// equal token — it is the only text in a fingerprint and the only thing
|
|
265
|
+
// that distinguishes two screens of the same shape. `fingerprint.js` has
|
|
266
|
+
// said so in a comment since it was written: "two list screens with
|
|
267
|
+
// identical structure differ by their title, and nothing else says so".
|
|
268
|
+
// Nothing enforced it.
|
|
269
|
+
//
|
|
270
|
+
// Measured on the RN testbed, which is what found this: a list of plants
|
|
271
|
+
// and a list of forms, each a large title over full-width rows above a
|
|
272
|
+
// tab bar, scored **0.50** against a 0.36 threshold and became one node.
|
|
273
|
+
// Both were correctly named by then; the name was simply outvoted, being
|
|
274
|
+
// one token of a union of six. The graph then offered one screen's
|
|
275
|
+
// controls on the other, which is the failure `verify` exists to stop.
|
|
276
|
+
//
|
|
277
|
+
// Both sides must actually carry names for this to apply. A reading whose
|
|
278
|
+
// tree did not answer has no names through no fault of the screen's, and
|
|
279
|
+
// the two sensors are already known to disagree about 0.33-0.47 of a
|
|
280
|
+
// token set — so "one has names, the other does not" is a fact about the
|
|
281
|
+
// sensors and must not be read as a fact about identity.
|
|
282
|
+
if (disagreeOnName(f.tokens, key.tokens)) continue;
|
|
261
283
|
const s = fingerprint.similarity(f.tokens, key.tokens);
|
|
262
284
|
if (s > bestSimilarity) {
|
|
263
285
|
bestSimilarity = s;
|
|
@@ -268,6 +290,31 @@ export function nearestScreen(udid, screen, { threshold = SIMILARITY_THRESHOLD }
|
|
|
268
290
|
return best && bestSimilarity >= threshold ? { node: best, similarity: bestSimilarity } : null;
|
|
269
291
|
}
|
|
270
292
|
|
|
293
|
+
/** The chrome labels in a token set — the only text a fingerprint keeps. */
|
|
294
|
+
function namesIn(tokens) {
|
|
295
|
+
const out = new Set();
|
|
296
|
+
for (const t of tokens ?? []) {
|
|
297
|
+
const open = t.indexOf('"');
|
|
298
|
+
if (open < 0) continue;
|
|
299
|
+
const close = t.lastIndexOf('"');
|
|
300
|
+
if (close > open) out.add(t.slice(open + 1, close));
|
|
301
|
+
}
|
|
302
|
+
return out;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/**
|
|
306
|
+
* Do these two readings name themselves, and name themselves differently?
|
|
307
|
+
*
|
|
308
|
+
* Only a positive disagreement counts. Silence on either side is not evidence.
|
|
309
|
+
*/
|
|
310
|
+
function disagreeOnName(a, b) {
|
|
311
|
+
const left = namesIn(a);
|
|
312
|
+
const right = namesIn(b);
|
|
313
|
+
if (!left.size || !right.size) return false;
|
|
314
|
+
for (const name of left) if (right.has(name)) return false;
|
|
315
|
+
return true;
|
|
316
|
+
}
|
|
317
|
+
|
|
271
318
|
/**
|
|
272
319
|
* Teach a node that it also looks like this.
|
|
273
320
|
*
|
|
@@ -295,6 +342,38 @@ function replayable(step) {
|
|
|
295
342
|
return step.launch != null || step.openUrl != null ? null : step;
|
|
296
343
|
}
|
|
297
344
|
|
|
345
|
+
/**
|
|
346
|
+
* What has worked from this screen before, in the caller's own vocabulary.
|
|
347
|
+
*
|
|
348
|
+
* The graph has always known this and never said it. A map reported
|
|
349
|
+
* `(known, 3 known exits)` — the *count* — so an agent on a screen simframe had
|
|
350
|
+
* driven successfully six times still had to read it to learn what was tappable.
|
|
351
|
+
* Measured across two peer rounds: a flow whose steps were known in advance ran
|
|
352
|
+
* 16 steps in **one** call, and the same agent on a screen the graph also knew
|
|
353
|
+
* but whose labels it did not spent 25 calls on 31 steps. The difference was not
|
|
354
|
+
* perception. It was whether a plan existed before execution started.
|
|
355
|
+
*
|
|
356
|
+
* Ordered by how often each has worked, because that is the order an agent
|
|
357
|
+
* should try them in.
|
|
358
|
+
*/
|
|
359
|
+
export function exitsOf(node, { limit = 8 } = {}) {
|
|
360
|
+
return (node?.edges ?? [])
|
|
361
|
+
.map((e) => ({
|
|
362
|
+
action: e.step?.action ?? (e.action ?? '').split(':')[0] ?? 'tap',
|
|
363
|
+
label: e.step?.value ?? e.step?.target ?? e.step?.label ?? e.step?.into ?? null,
|
|
364
|
+
to: e.to ?? null,
|
|
365
|
+
count: e.count ?? 0,
|
|
366
|
+
kind: e.kind ?? null,
|
|
367
|
+
}))
|
|
368
|
+
// A `#13` was a ref on the screen it was typed on and means nothing on the
|
|
369
|
+
// next visit; a raw coordinate is not a name either. Neither is reusable
|
|
370
|
+
// vocabulary, which is the whole point of this list.
|
|
371
|
+
.filter((e) => e.label != null && !/^#\d+$/.test(String(e.label).trim())
|
|
372
|
+
&& !/^@?-?\d+\s*,\s*-?\d+$/.test(String(e.label).trim()))
|
|
373
|
+
.sort((a, b) => b.count - a.count)
|
|
374
|
+
.slice(0, limit);
|
|
375
|
+
}
|
|
376
|
+
|
|
298
377
|
/**
|
|
299
378
|
* What to call this screen, for a human typing `goto`.
|
|
300
379
|
*
|
|
@@ -756,12 +835,36 @@ export const VERDICTS = ['ok', 'no-visible-change', 'unexpected-screen', 'unveri
|
|
|
756
835
|
* app and the verdict still said `unexpected-screen`, because the verdict never
|
|
757
836
|
* asked the graph.
|
|
758
837
|
*/
|
|
838
|
+
/**
|
|
839
|
+
* Are these two readings the same screen?
|
|
840
|
+
*
|
|
841
|
+
* `nearestScreen` has always had a token-similarity tolerance, precisely so
|
|
842
|
+
* that a screen whose *content* differs — a list with different rows, a form
|
|
843
|
+
* showing a different record — still resolves to the screen it is. The
|
|
844
|
+
* verification path threw that away: it passed bare hash strings, and a string
|
|
845
|
+
* carries no tokens, so only an exact hash could ever match.
|
|
846
|
+
*
|
|
847
|
+
* The cost was measured. `unexpected-screen` fired three times in one reported
|
|
848
|
+
* run and was wrong all three; two were this — the tester picked a different
|
|
849
|
+
* asset than earlier runs had, so the content differed, so the hash differed,
|
|
850
|
+
* so a correct navigation was called a wrong turn. Their conclusion: *"this
|
|
851
|
+
* will fire on every run that varies its test data — i.e. every useful run."*
|
|
852
|
+
* And because a failed step abandons the rest of its batch, each false alarm
|
|
853
|
+
* costs a round trip, which is the thing the whole design is trying to buy.
|
|
854
|
+
*
|
|
855
|
+
* So pass the reading, not just its name: `{hash, tokens}` lets the tolerance
|
|
856
|
+
* that already exists do its job. A string still works and still means "exact
|
|
857
|
+
* match only", which is right for a stored prediction that has no tokens.
|
|
858
|
+
*/
|
|
759
859
|
function sameScreen(udid, a, b) {
|
|
760
|
-
|
|
761
|
-
|
|
860
|
+
const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
|
|
861
|
+
const ha = hashOf(a);
|
|
862
|
+
const hb = hashOf(b);
|
|
863
|
+
if (!ha || !hb) return false;
|
|
864
|
+
if (ha === hb) return true;
|
|
762
865
|
if (!udid) return false;
|
|
763
|
-
const nodeA = nearestScreen(udid, a)?.node;
|
|
764
|
-
const nodeB = nearestScreen(udid, b)?.node;
|
|
866
|
+
const nodeA = nearestScreen(udid, typeof a === 'string' ? a : { hash: ha, tokens: a?.tokens })?.node;
|
|
867
|
+
const nodeB = nearestScreen(udid, typeof b === 'string' ? b : { hash: hb, tokens: b?.tokens })?.node;
|
|
765
868
|
return Boolean(nodeA && nodeB && nodeA.hash === nodeB.hash);
|
|
766
869
|
}
|
|
767
870
|
|
|
@@ -777,9 +880,35 @@ function sameScreen(udid, a, b) {
|
|
|
777
880
|
*/
|
|
778
881
|
export const CONFIDENT_OBSERVATIONS = 2;
|
|
779
882
|
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
883
|
+
/**
|
|
884
|
+
* Actions whose correct outcome is that the screen stays where it is.
|
|
885
|
+
*
|
|
886
|
+
* Typing into a field does not navigate, so screen-identity movement cannot
|
|
887
|
+
* say whether it worked — and answering with `no-visible-change` was actively
|
|
888
|
+
* harmful three ways. It printed a verdict contradicting the wait's own
|
|
889
|
+
* observation on the same line (`[a small change, in one region only] …
|
|
890
|
+
* [no-visible-change]`, both true, of different questions). It is an escalating
|
|
891
|
+
* verdict, so a clean flow that typed anything told the caller to stop and
|
|
892
|
+
* think. And it invited a re-type, which doubles a field that cannot be
|
|
893
|
+
* cleared.
|
|
894
|
+
*
|
|
895
|
+
* These steps are verified by reading the field back instead — see
|
|
896
|
+
* `fieldContents` in actions.js.
|
|
897
|
+
*/
|
|
898
|
+
export const STAYS_ON_SCREEN = new Set(['type', 'paste', 'key']);
|
|
899
|
+
|
|
900
|
+
export function verdict({ udid, prediction, before, after, kind, action }) {
|
|
901
|
+
// `before`/`after` may be a hash or a whole reading. A reading carries its
|
|
902
|
+
// tokens, which is what lets a content-varied screen still be recognised as
|
|
903
|
+
// the screen it is.
|
|
904
|
+
const hashOf = (v) => (typeof v === 'string' ? v : v?.hash);
|
|
905
|
+
const beforeHash = hashOf(before);
|
|
906
|
+
const afterHash = hashOf(after);
|
|
907
|
+
if (!beforeHash || !afterHash) return { verdict: 'unverified', detail: 'no state to compare' };
|
|
908
|
+
const moved = beforeHash !== afterHash;
|
|
909
|
+
if (STAYS_ON_SCREEN.has(action) && !moved) {
|
|
910
|
+
return { verdict: 'ok', detail: 'the screen was not expected to change, and did not' };
|
|
911
|
+
}
|
|
783
912
|
if (!prediction) {
|
|
784
913
|
if (!moved) return { verdict: 'no-visible-change', detail: 'the screen did not change, and nothing predicted it would' };
|
|
785
914
|
return { verdict: 'unverified', detail: 'this action has not been seen on this screen before' };
|