simframe 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +137 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1692 -44
- package/src/cli.js +131 -9
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +89 -7
- package/src/index.js +211 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +155 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +319 -27
- package/src/metrics.js +32 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +2 -1
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +117 -0
- package/src/view.js +335 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/index.js
CHANGED
|
@@ -5,6 +5,7 @@ import path from 'node:path';
|
|
|
5
5
|
import { fileURLToPath } from 'node:url';
|
|
6
6
|
import { DEFAULTS, STATE_VERSION } from './daemon.js';
|
|
7
7
|
import * as engine from './engine.js';
|
|
8
|
+
import * as regions from './regions.js';
|
|
8
9
|
import { decodePng, encodePng, scaleBitmap } from './png.js';
|
|
9
10
|
import {
|
|
10
11
|
REGION_COLS,
|
|
@@ -536,9 +537,31 @@ export async function getFrame(deviceQuery, { detail = 'normal', options } = {})
|
|
|
536
537
|
let scaledOnRead = false;
|
|
537
538
|
if (maxDim === 0) {
|
|
538
539
|
file = state.fullFile;
|
|
539
|
-
} else if (maxDim > nativeMax + 8
|
|
540
|
+
} else if (maxDim > nativeMax + 8) {
|
|
541
|
+
// Asking for more detail than the ring frame holds, so it has to come from
|
|
542
|
+
// a full-resolution frame. The `fs.existsSync(state.fullFile)` this used to
|
|
543
|
+
// require is exactly the condition that fails routinely — retention thins
|
|
544
|
+
// full frames aggressively, and `state.fullFile` names one that is often
|
|
545
|
+
// already gone. When it did, this fell straight through to `latest.png` and
|
|
546
|
+
// returned a 322x700 image while still calling itself detail "high".
|
|
547
|
+
//
|
|
548
|
+
// That was not a small inaccuracy. Every single `sim_look` header in a
|
|
549
|
+
// two-agent field round read `322x700` on a 402x874pt device, so "high
|
|
550
|
+
// (1024px, readable small text)" was interpolating a sub-1x source the
|
|
551
|
+
// whole time. `fullFrameFor` has always known how to find or take a full
|
|
552
|
+
// frame — state named full/9660.png while the directory held eight frames
|
|
553
|
+
// at 1206x2622 — and this path simply never asked it.
|
|
554
|
+
//
|
|
555
|
+
// What this does *not* explain, though it is tempting: the OCR corruption
|
|
556
|
+
// in that round's log (`saaiad@example.com`, `suomit order`). OCR has
|
|
557
|
+
// always gone through `fullFrameFor`, and the daemon's own text recognition
|
|
558
|
+
// reads the live surface at native resolution, so neither was ever looking
|
|
559
|
+
// at the small frame. Those errors are a desktop-width page rendering its
|
|
560
|
+
// labels at a few pixels tall. Fixing what the caller *sees* does not fix
|
|
561
|
+
// what OCR reads, and saying so here keeps the next reader from assuming it did.
|
|
562
|
+
const source = await fullFrameFor(device.udid, state);
|
|
540
563
|
const out = path.join(p.dir, `read-${maxDim}.png`);
|
|
541
|
-
await resize(
|
|
564
|
+
await resize(source, out, maxDim);
|
|
542
565
|
file = out;
|
|
543
566
|
scaledOnRead = true;
|
|
544
567
|
} else if (maxDim < nativeMax - 8) {
|
|
@@ -550,17 +573,50 @@ export async function getFrame(deviceQuery, { detail = 'normal', options } = {})
|
|
|
550
573
|
|
|
551
574
|
const png = fs.readFileSync(file);
|
|
552
575
|
const bmp = pngSize(png);
|
|
576
|
+
// The age must describe the *bytes*, not the state that dates them.
|
|
577
|
+
//
|
|
578
|
+
// These are two different files. `state.json` is written when a frame is
|
|
579
|
+
// recorded; the image is a separate write. A freshly started daemon publishes
|
|
580
|
+
// fresh state while `latest.png` is still the previous session's — which is
|
|
581
|
+
// when a caller's first look of a session happens. Reported, and it is the
|
|
582
|
+
// worst failure this tool can have: a header reading `frame #714 · 84ms old ·
|
|
583
|
+
// still for 17173ms` above an image whose status-bar clock said 6:11 when the
|
|
584
|
+
// real time was 11:10. `sim_ui` was right in the same session because the
|
|
585
|
+
// accessibility tree is read live and in-process; only the image comes from a
|
|
586
|
+
// file, so only the image can be hours stale while the header says otherwise.
|
|
587
|
+
//
|
|
588
|
+
// Taking the *larger* of the two ages cannot overstate freshness. It can
|
|
589
|
+
// overstate staleness by however long the two writes are apart, which is 6ms
|
|
590
|
+
// measured on a healthy daemon — the right direction to be wrong in.
|
|
591
|
+
let fileAgeMs = null;
|
|
592
|
+
try { fileAgeMs = Date.now() - fs.statSync(file).mtimeMs; } catch { /* stat is advisory */ }
|
|
593
|
+
const stateAgeMs = Date.now() - state.capturedAt;
|
|
594
|
+
const ageMs = Number.isFinite(fileAgeMs) ? Math.max(stateAgeMs, fileAgeMs) : stateAgeMs;
|
|
553
595
|
return {
|
|
554
596
|
device,
|
|
555
597
|
state,
|
|
556
598
|
png,
|
|
557
599
|
width: bmp.width,
|
|
558
600
|
height: bmp.height,
|
|
559
|
-
ageMs
|
|
601
|
+
ageMs,
|
|
602
|
+
// Said out loud when the image is materially older than the state, because
|
|
603
|
+
// "this picture is not of the screen the rest of this response describes" is
|
|
604
|
+
// not something a caller can work out for themselves.
|
|
605
|
+
frameBehindMs: Number.isFinite(fileAgeMs) && fileAgeMs - stateAgeMs > FRAME_BEHIND_MS
|
|
606
|
+
? Math.round(fileAgeMs - stateAgeMs)
|
|
607
|
+
: null,
|
|
560
608
|
scaledOnRead,
|
|
561
609
|
};
|
|
562
610
|
}
|
|
563
611
|
|
|
612
|
+
/**
|
|
613
|
+
* How far the image may lag the state before it is worth saying so.
|
|
614
|
+
*
|
|
615
|
+
* Measured on a healthy daemon, the two writes land 6ms apart. A second is far
|
|
616
|
+
* outside that and far inside the hours-stale case this exists to catch.
|
|
617
|
+
*/
|
|
618
|
+
const FRAME_BEHIND_MS = 1000;
|
|
619
|
+
|
|
564
620
|
function pngSize(png) {
|
|
565
621
|
return { width: png.readUInt32BE(16), height: png.readUInt32BE(20) };
|
|
566
622
|
}
|
|
@@ -1068,10 +1124,76 @@ export async function readScreenWith(deviceQuery, { useAx = true, useOcr = true,
|
|
|
1068
1124
|
return { device, entry, points: { width: geo.pointWidth, height: geo.pointHeight } };
|
|
1069
1125
|
}
|
|
1070
1126
|
|
|
1071
|
-
|
|
1127
|
+
/**
|
|
1128
|
+
* A target that answers the query but sits outside the viewport.
|
|
1129
|
+
*
|
|
1130
|
+
* The off-screen filter above is right — an element in the tree below the fold
|
|
1131
|
+
* is untappable in fact — but "not on this screen" was the wrong way to say so.
|
|
1132
|
+
* Reported: a `waitFor` spent its full 15s timeout and stopped the flow while
|
|
1133
|
+
* the control sat one scroll down. Waiting cannot fix that and scrolling can.
|
|
1134
|
+
*/
|
|
1135
|
+
export function offScreenMatch(targets, query, points) {
|
|
1136
|
+
// Both axes: a horizontal row puts elements past the right edge, and checking
|
|
1137
|
+
// only `y` reported them as visible.
|
|
1138
|
+
const off = (targets ?? []).filter((t) => t.label && regions.offViewport(t, points));
|
|
1139
|
+
if (!off.length) return null;
|
|
1140
|
+
const hit = matching.resolve(off, query);
|
|
1141
|
+
if (hit.status === 'ok') return hit.target;
|
|
1142
|
+
// Ambiguous off-screen is still an answer to "why is it not here".
|
|
1143
|
+
return hit.status === 'ambiguous' && hit.alternatives?.length ? hit.alternatives[0] : null;
|
|
1144
|
+
}
|
|
1145
|
+
|
|
1146
|
+
/**
|
|
1147
|
+
* Which sensors a read asks for by default.
|
|
1148
|
+
*
|
|
1149
|
+
* `full` is the default and what CLAUDE.md fixes: accessibility and OCR fused
|
|
1150
|
+
* into one element list. `ax-first` asks for the tree alone — 85 ms against
|
|
1151
|
+
* 142 ms warm — and pays for OCR only when the cheap read could not answer the
|
|
1152
|
+
* question.
|
|
1153
|
+
*
|
|
1154
|
+
* The escalation is the whole point, and it is what makes this safe to try. A
|
|
1155
|
+
* mode that just dropped OCR would lose every OCR-only element, which is
|
|
1156
|
+
* precisely how the map cut lost discovery: nothing became untappable, but the
|
|
1157
|
+
* agent could no longer see what was there. Here a resolve failure — the one
|
|
1158
|
+
* signal that says "the cheap sensor was not enough" — triggers a full read
|
|
1159
|
+
* before anyone is told the target is absent. Wrong guesses cost a second read;
|
|
1160
|
+
* they cannot cost a wrong answer.
|
|
1161
|
+
*/
|
|
1162
|
+
export function sensorMode(options) {
|
|
1163
|
+
// Per call first, then the environment. The environment is fixed when a
|
|
1164
|
+
// process starts, and an MCP server is one long-lived process — so a tester
|
|
1165
|
+
// asked to compare two sensor modes in one session could not do it, which is
|
|
1166
|
+
// exactly what happened: round 6's A was run and B and C could not be. Their
|
|
1167
|
+
// own suggestion was three server entries with three env blocks, and it is
|
|
1168
|
+
// the worse fix: three servers on one device means three writers, against the
|
|
1169
|
+
// one-writer-per-device rule, and it makes an A/B a configuration change
|
|
1170
|
+
// rather than an argument.
|
|
1171
|
+
const asked = options?.sensor;
|
|
1172
|
+
const raw = String(asked ?? process.env.SIMFRAME_SENSOR ?? '').trim().toLowerCase();
|
|
1173
|
+
return raw === 'ax-first' || raw === 'axfirst' ? 'ax-first' : 'full';
|
|
1174
|
+
}
|
|
1175
|
+
|
|
1176
|
+
export async function locate(deviceQuery, query, opts = {}) {
|
|
1177
|
+
if (sensorMode(opts.options) !== 'ax-first' || opts.useOcr === false || opts.escalated) {
|
|
1178
|
+
return locateWith(deviceQuery, query, opts);
|
|
1179
|
+
}
|
|
1180
|
+
const options = opts;
|
|
1181
|
+
try {
|
|
1182
|
+
return await locateWith(deviceQuery, query, { ...options, useOcr: false, escalated: true });
|
|
1183
|
+
} catch (err) {
|
|
1184
|
+
// Only a perception failure earns the expensive retry. A refused selector or
|
|
1185
|
+
// an ambiguity between two things the tree *did* see is not going to be
|
|
1186
|
+
// settled by reading more text.
|
|
1187
|
+
const why = metrics.escalationOf(err);
|
|
1188
|
+
if (why?.reason !== 'unknown_screen' && why?.reason !== 'ambiguous_intent') throw err;
|
|
1189
|
+
return locateWith(deviceQuery, query, { ...options, useOcr: true, refresh: true, escalated: true });
|
|
1190
|
+
}
|
|
1191
|
+
}
|
|
1192
|
+
|
|
1193
|
+
async function locateWith(
|
|
1072
1194
|
deviceQuery,
|
|
1073
1195
|
query,
|
|
1074
|
-
{ index, refresh = false, useAx = true, useOcr = true, settleMs = MEMORY_SETTLE_MS, options } = {},
|
|
1196
|
+
{ index, refresh = false, useAx = true, useOcr = true, settleMs = MEMORY_SETTLE_MS, options, escalated } = {},
|
|
1075
1197
|
) {
|
|
1076
1198
|
const { device, state: firstState } = await ensureDaemon(deviceQuery, options);
|
|
1077
1199
|
const udid = device.udid;
|
|
@@ -1094,11 +1216,47 @@ export async function locate(
|
|
|
1094
1216
|
// be checked against structural identity without paying for a perception
|
|
1095
1217
|
// pass — which is the whole reason a ref exists.
|
|
1096
1218
|
const near = screenmap.recallNearest(udid, firstState.layoutHash);
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1219
|
+
let hit;
|
|
1220
|
+
try {
|
|
1221
|
+
hit = resolveRefHere();
|
|
1222
|
+
} catch (err) {
|
|
1223
|
+
// A stale ref carries the label it was numbered against, so it need not
|
|
1224
|
+
// cost the rest of the batch. Reported: `#19 was numbered on a different
|
|
1225
|
+
// screen (61b835b7 → 6f34c006)` because dashboard cards finished loading
|
|
1226
|
+
// and shifted the layout — the same screen, a new hash — and that one
|
|
1227
|
+
// refusal aborted the three remaining steps.
|
|
1228
|
+
//
|
|
1229
|
+
// It re-resolves by label rather than by coordinate, and it *says so*.
|
|
1230
|
+
// The number is not honoured; the caller's own words are, which is what
|
|
1231
|
+
// they would have written instead. Resolving the old coordinates would be
|
|
1232
|
+
// the dangerous version of this, and is not what happens.
|
|
1233
|
+
// Only layout drift is recoverable. `identity` means the numbers were
|
|
1234
|
+
// drawn somewhere else, and the label they stood for appearing here is
|
|
1235
|
+
// coincidence rather than evidence — measured: `#1` numbered "Reminders"
|
|
1236
|
+
// in Reminders re-resolved in Contacts onto the status-bar back-to-app
|
|
1237
|
+
// breadcrumb "• Reminders", and reported ok.
|
|
1238
|
+
if (!err.staleRef || !err.staleLabel || err.staleKind !== 'drift') throw err;
|
|
1239
|
+
let again;
|
|
1240
|
+
try {
|
|
1241
|
+
again = await locateWith(deviceQuery, err.staleLabel, {
|
|
1242
|
+
index, refresh: true, useAx, useOcr, settleMs, options, escalated,
|
|
1243
|
+
});
|
|
1244
|
+
} catch (second) {
|
|
1245
|
+
// The number was not honoured and the label it stood for is not here
|
|
1246
|
+
// either. That is still a refusal to tap stale coordinates, but by the
|
|
1247
|
+
// time it surfaces it looks like an ordinary "not on this screen" and a
|
|
1248
|
+
// caller cannot tell the two apart. Carrying the marker across says
|
|
1249
|
+
// which question was actually asked.
|
|
1250
|
+
second.staleRef = true;
|
|
1251
|
+
second.staleLabel = err.staleLabel;
|
|
1252
|
+
throw second;
|
|
1253
|
+
}
|
|
1254
|
+
return {
|
|
1255
|
+
...again,
|
|
1256
|
+
from: 'ref-relabelled',
|
|
1257
|
+
relabelled: { ref: selector.ref, label: err.staleLabel, why: err.message },
|
|
1258
|
+
};
|
|
1259
|
+
}
|
|
1102
1260
|
return {
|
|
1103
1261
|
device,
|
|
1104
1262
|
state: firstState,
|
|
@@ -1107,6 +1265,23 @@ export async function locate(
|
|
|
1107
1265
|
distance: 0,
|
|
1108
1266
|
settled: true,
|
|
1109
1267
|
};
|
|
1268
|
+
|
|
1269
|
+
function resolveRefHere() {
|
|
1270
|
+
return refs.resolveRef(udid, selector.ref, {
|
|
1271
|
+
layoutHash: firstState.layoutHash,
|
|
1272
|
+
structuralHash: near?.entry?.structuralHash ?? null,
|
|
1273
|
+
// How far that recall reached. `recallNearest` is deliberately tolerant —
|
|
1274
|
+
// a list with new rows is still the same screen — so at any distance
|
|
1275
|
+
// above zero it is a *guess* about which screen this is, and a guess must
|
|
1276
|
+
// not be the sole grounds for refusing a ref. Reported from the field: a
|
|
1277
|
+
// refusal reading `#5 was numbered on a different screen
|
|
1278
|
+
// (0f7b9e3e → 48e5c92d)` where both calls' headers printed the identical
|
|
1279
|
+
// screen, because the map named the screen from a tolerant recall while
|
|
1280
|
+
// refs treated that same recall as exact.
|
|
1281
|
+
structuralDistance: near?.distance ?? null,
|
|
1282
|
+
screenKnown: Boolean(near),
|
|
1283
|
+
});
|
|
1284
|
+
}
|
|
1110
1285
|
}
|
|
1111
1286
|
if (selector.exact) query = selector.label;
|
|
1112
1287
|
// Key memory off a settled frame, never off whichever frame happened to be
|
|
@@ -1167,7 +1342,7 @@ export async function locate(
|
|
|
1167
1342
|
// ambiguous_intent when the screen was one we thought we knew. A
|
|
1168
1343
|
// waiting caller needs the difference: more time cannot make a thing
|
|
1169
1344
|
// unique, and it can make an absent thing arrive.
|
|
1170
|
-
{ candidates: outcome.alternatives, ambiguous: true },
|
|
1345
|
+
{ candidates: outcome.alternatives, ambiguous: true, intent: query },
|
|
1171
1346
|
);
|
|
1172
1347
|
}
|
|
1173
1348
|
if (outcome.status === 'ok') {
|
|
@@ -1181,8 +1356,25 @@ export async function locate(
|
|
|
1181
1356
|
// here undoes every guard above — it has no off-screen filter and no
|
|
1182
1357
|
// coverage weighting, and it is what returned a scrolled-away list row for
|
|
1183
1358
|
// "back". "Not found" is the correct answer.
|
|
1184
|
-
const visible = entry.targets.filter((t) => t.label &&
|
|
1359
|
+
const visible = entry.targets.filter((t) => t.label && !regions.offViewport(t, points));
|
|
1185
1360
|
const sample = visible.slice(0, 12).map((t) => t.label.slice(0, 24)).join(', ');
|
|
1361
|
+
// "Not on this screen" and "not in view" are different answers, and giving
|
|
1362
|
+
// the first for the second cost a reported 15 seconds: a `waitFor REVIEW`
|
|
1363
|
+
// burned its whole timeout while REVIEW sat one scroll below the fold, and
|
|
1364
|
+
// then stopped the flow. Waiting cannot bring a thing into view, and
|
|
1365
|
+
// scrolling can — so the difference is the whole of what to do next.
|
|
1366
|
+
const offScreen = offScreenMatch(entry.targets, query, points);
|
|
1367
|
+
if (offScreen) {
|
|
1368
|
+
throw metrics.tag(
|
|
1369
|
+
new Error(
|
|
1370
|
+
`"${query}" is in the tree but not in view — it is at y=${Math.round(offScreen.y)}`
|
|
1371
|
+
+ ` on a ${Math.round(points.height)}pt screen. Scroll to it (sim_scroll_to) rather than waiting;`
|
|
1372
|
+
+ ' waiting cannot bring it into view.',
|
|
1373
|
+
),
|
|
1374
|
+
from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',
|
|
1375
|
+
{ candidates: visible.slice(0, 8), intent: query },
|
|
1376
|
+
);
|
|
1377
|
+
}
|
|
1186
1378
|
// Which escalation this is depends on whether the screen was recognised.
|
|
1187
1379
|
// Screen memory had nothing for it (`from` is one of the built values) and
|
|
1188
1380
|
// the target is missing: that is not knowing the screen. On a screen
|
|
@@ -1190,7 +1382,7 @@ export async function locate(
|
|
|
1190
1382
|
throw metrics.tag(
|
|
1191
1383
|
new Error(`"${query}" is not on this screen. Visible: ${sample || '(nothing readable)'}`),
|
|
1192
1384
|
from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',
|
|
1193
|
-
{ candidates: visible.slice(0, 8) },
|
|
1385
|
+
{ candidates: visible.slice(0, 8), intent: query },
|
|
1194
1386
|
);
|
|
1195
1387
|
}
|
|
1196
1388
|
|
|
@@ -1202,7 +1394,7 @@ export async function locate(
|
|
|
1202
1394
|
throw metrics.tag(
|
|
1203
1395
|
new Error(`"${query}" is not on this screen. Visible: ${sample || '(nothing readable)'}`),
|
|
1204
1396
|
from === 'memory' ? 'ambiguous_intent' : 'unknown_screen',
|
|
1205
|
-
{ candidates: visible.slice(0, 8) },
|
|
1397
|
+
{ candidates: visible.slice(0, 8), intent: query },
|
|
1206
1398
|
);
|
|
1207
1399
|
}
|
|
1208
1400
|
return { device, state: current, entry, target, from, distance, settled, screens: screenmap.stats(udid).screens };
|
|
@@ -1315,6 +1507,11 @@ export async function screenIdentity(deviceQuery, { options, confirmNovel = true
|
|
|
1315
1507
|
keyboard: Boolean(entry.keyboard),
|
|
1316
1508
|
layoutHash: current.layoutHash,
|
|
1317
1509
|
settled,
|
|
1510
|
+
// Settled and incomplete are different states and used to render
|
|
1511
|
+
// identically. A screen awaiting a network call is perfectly still; a
|
|
1512
|
+
// person sees a spinner and knows to wait. The classifier already says
|
|
1513
|
+
// so, and nothing above this line was asking.
|
|
1514
|
+
loading: current.transition?.kind === 'loading',
|
|
1318
1515
|
// Carried out so callers that want the elements as well as the identity
|
|
1319
1516
|
// do not pay for a second perception pass to get them. The compact
|
|
1320
1517
|
// screen map needs both, and reading twice was the whole cost of it.
|
package/src/input.js
CHANGED
|
@@ -429,14 +429,77 @@ export async function typeKeys(udid, value) {
|
|
|
429
429
|
await idb(['ui', 'text', '--udid', udid, String(value)]);
|
|
430
430
|
}
|
|
431
431
|
|
|
432
|
-
|
|
432
|
+
/**
|
|
433
|
+
* Keyboard keys, by name.
|
|
434
|
+
*
|
|
435
|
+
* A peer was blocked outright for want of Return: half of mobile search fields
|
|
436
|
+
* submit on the keyboard return key, `button` covers only the hardware buttons,
|
|
437
|
+
* and `key` wanted a raw HID usage code that nobody should have to know. Typing
|
|
438
|
+
* "\n" as text is not a substitute — text goes through whatever keyboard layout
|
|
439
|
+
* iOS has active, and measured, it turned "Coke Display" into "Coke In Display".
|
|
440
|
+
*
|
|
441
|
+
* These are HID keyboard usage codes, which name a key *position* and are never
|
|
442
|
+
* translated by a layout. That property is the whole reason this path exists on
|
|
443
|
+
* a device whose own doctor warns that two extra layouts are installed.
|
|
444
|
+
*/
|
|
445
|
+
export const KEYS = {
|
|
446
|
+
return: 40, enter: 40, escape: 41, esc: 41, backspace: 42, delete: 42,
|
|
447
|
+
tab: 43, space: 44, up: 82, down: 81, left: 80, right: 79,
|
|
448
|
+
a: 4,
|
|
449
|
+
};
|
|
450
|
+
|
|
451
|
+
/** Modifier usage codes, held while another key is pressed. */
|
|
452
|
+
export const MODIFIERS = { control: 224, shift: 225, alt: 226, option: 226, command: 227, cmd: 227, gui: 227 };
|
|
453
|
+
|
|
454
|
+
/** The usage code for a name, a number, or null when it is neither. */
|
|
455
|
+
export function keyUsage(key) {
|
|
456
|
+
if (Number.isFinite(Number(key))) return Number(key);
|
|
457
|
+
const name = String(key ?? '').trim().toLowerCase();
|
|
458
|
+
return Object.hasOwn(KEYS, name) ? KEYS[name] : null;
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
export async function pressKey(udid, keycode, { modifiers = [] } = {}) {
|
|
433
462
|
await ensureFreshSession(udid);
|
|
463
|
+
const usage = keyUsage(keycode);
|
|
464
|
+
const held = modifiers
|
|
465
|
+
.map((m) => (Number.isFinite(Number(m)) ? Number(m) : MODIFIERS[String(m).trim().toLowerCase()]))
|
|
466
|
+
.filter((m) => Number.isFinite(m));
|
|
467
|
+
if (usage == null) {
|
|
468
|
+
throw new Error(`unknown key ${JSON.stringify(String(keycode))} — known names: ${Object.keys(KEYS).join(', ')}`
|
|
469
|
+
+ ', or a HID usage code');
|
|
470
|
+
}
|
|
434
471
|
const own = inputDriverFor(udid);
|
|
435
472
|
if (own) {
|
|
436
|
-
await own.key(udid,
|
|
473
|
+
await own.key(udid, usage, held);
|
|
474
|
+
return;
|
|
475
|
+
}
|
|
476
|
+
// The daemon owns the keyboard usage path on iOS. It was implemented in the
|
|
477
|
+
// HID layer and never exposed as a verb, so this fell through to idb — which
|
|
478
|
+
// is absent on a machine using the daemon, and so there was no way to press a
|
|
479
|
+
// keyboard key at all.
|
|
480
|
+
if (control.available(udid)) {
|
|
481
|
+
await control.key(udid, usage, held);
|
|
437
482
|
return;
|
|
438
483
|
}
|
|
439
|
-
|
|
484
|
+
if (held.length) throw new Error('modifier keys need the daemon; idb cannot hold one');
|
|
485
|
+
await idb(['ui', 'key', '--udid', udid, String(usage)]);
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
/**
|
|
489
|
+
* Empty the focused field.
|
|
490
|
+
*
|
|
491
|
+
* Command-A then Delete, over HID. There is no clear primitive anywhere —
|
|
492
|
+
* XCUITest, Appium and idb all lack one, and re-typing appends — so this is the
|
|
493
|
+
* standard answer rather than a trick of ours. It is layout-independent for the
|
|
494
|
+
* reason that matters on a device with Farsi and Armenian keyboards installed:
|
|
495
|
+
* a modifier and Delete are key *positions*, and so is the `a` in Command-A, so
|
|
496
|
+
* none of the three is translated by the active layout.
|
|
497
|
+
*
|
|
498
|
+
* It clears whatever has focus, which is why every caller focuses first.
|
|
499
|
+
*/
|
|
500
|
+
export async function clearField(udid) {
|
|
501
|
+
await pressKey(udid, 'a', { modifiers: ['command'] });
|
|
502
|
+
await pressKey(udid, 'delete');
|
|
440
503
|
}
|
|
441
504
|
|
|
442
505
|
/**
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A warm, line-oriented local helper process.
|
|
3
|
+
*
|
|
4
|
+
* Extracted rather than duplicated, because writing this twice would mean
|
|
5
|
+
* risking the same two bugs twice — and both were subtle enough to look like
|
|
6
|
+
* something else entirely.
|
|
7
|
+
*
|
|
8
|
+
* A timed-out request left its waiter in the queue, so every later answer went
|
|
9
|
+
* to the wrong asker and the run simply never finished; it read as the model
|
|
10
|
+
* being slow. And unreferencing the child's stdout unreferenced the pipe every
|
|
11
|
+
* request waits on, so the process exited silently in the middle of an await
|
|
12
|
+
* and printed nothing at all, returning 0.
|
|
13
|
+
*
|
|
14
|
+
* The helper is kept warm because the first answer in a process pays model load
|
|
15
|
+
* — measured at ~880ms against ~560ms for every answer after it.
|
|
16
|
+
*/
|
|
17
|
+
import { spawn, execFile } from 'node:child_process';
|
|
18
|
+
import fs from 'node:fs';
|
|
19
|
+
import path from 'node:path';
|
|
20
|
+
import { promisify } from 'node:util';
|
|
21
|
+
|
|
22
|
+
const run = promisify(execFile);
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Compile a Swift source once and reuse the binary.
|
|
26
|
+
*
|
|
27
|
+
* `swiftc`, not `xcrun swiftc`: nothing above the platform boundary may name a
|
|
28
|
+
* platform tool, and the boundary test catches it. A compiler is not a device
|
|
29
|
+
* tool, which is the precedent `src/ocr.js` set.
|
|
30
|
+
*/
|
|
31
|
+
export function compiler({ source, binary, what }) {
|
|
32
|
+
let building = null;
|
|
33
|
+
return async function ensureBinary() {
|
|
34
|
+
if (building) return building;
|
|
35
|
+
building = (async () => {
|
|
36
|
+
try {
|
|
37
|
+
const src = fs.statSync(source).mtimeMs;
|
|
38
|
+
const bin = fs.existsSync(binary) ? fs.statSync(binary).mtimeMs : 0;
|
|
39
|
+
if (bin > src) return { available: true, binary };
|
|
40
|
+
} catch {
|
|
41
|
+
return { available: false, reason: `the ${what} source is missing from this install` };
|
|
42
|
+
}
|
|
43
|
+
try {
|
|
44
|
+
fs.mkdirSync(path.dirname(binary), { recursive: true });
|
|
45
|
+
await run('swiftc', ['-O', source, '-o', binary], { timeout: 180_000 });
|
|
46
|
+
return { available: true, binary };
|
|
47
|
+
} catch (err) {
|
|
48
|
+
building = null; // let a later call retry once a toolchain is present
|
|
49
|
+
return {
|
|
50
|
+
available: false,
|
|
51
|
+
reason: err.code === 'ENOENT'
|
|
52
|
+
? `swiftc is not installed, so the ${what} cannot be built (install Xcode command line tools)`
|
|
53
|
+
: `could not build the ${what}: ${String(err.message).split('\n')[0]}`,
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
})();
|
|
57
|
+
return building;
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Open a helper and speak JSON lines to it.
|
|
63
|
+
*
|
|
64
|
+
* @returns {{ask: (o: object, ms: number) => Promise<object|null>, close: () => void, ok: boolean, reason?: string}}
|
|
65
|
+
*/
|
|
66
|
+
export function lineServer({ ensureBinary, what }) {
|
|
67
|
+
let session = null;
|
|
68
|
+
|
|
69
|
+
async function open() {
|
|
70
|
+
if (session) return session;
|
|
71
|
+
const built = await ensureBinary();
|
|
72
|
+
if (!built.available) return { ok: false, reason: built.reason };
|
|
73
|
+
session = await new Promise((resolve) => {
|
|
74
|
+
const child = spawn(built.binary, [], { stdio: ['pipe', 'pipe', 'ignore'] });
|
|
75
|
+
// Deliberately NOT unref'd — see the note at the top of this file.
|
|
76
|
+
let buffer = '';
|
|
77
|
+
const waiters = [];
|
|
78
|
+
let settled = false;
|
|
79
|
+
const fail = (reason) => {
|
|
80
|
+
if (!settled) { settled = true; resolve({ ok: false, reason }); }
|
|
81
|
+
while (waiters.length) waiters.shift()(null);
|
|
82
|
+
};
|
|
83
|
+
child.on('error', (err) => fail(`the ${what} would not start: ${err.message}`));
|
|
84
|
+
child.on('exit', () => { session = null; fail(`the ${what} exited`); });
|
|
85
|
+
child.stdout.on('data', (chunk) => {
|
|
86
|
+
buffer += chunk;
|
|
87
|
+
let i = buffer.indexOf('\n');
|
|
88
|
+
while (i >= 0) {
|
|
89
|
+
const line = buffer.slice(0, i).trim();
|
|
90
|
+
buffer = buffer.slice(i + 1);
|
|
91
|
+
i = buffer.indexOf('\n');
|
|
92
|
+
if (!line) continue;
|
|
93
|
+
let msg;
|
|
94
|
+
try { msg = JSON.parse(line); } catch { continue; }
|
|
95
|
+
if (!settled) {
|
|
96
|
+
settled = true;
|
|
97
|
+
if (msg.ready) resolve({ ok: true, child, waiters });
|
|
98
|
+
else resolve({ ok: false, reason: msg.unavailable ?? `the ${what} did not become ready` });
|
|
99
|
+
continue;
|
|
100
|
+
}
|
|
101
|
+
const next = waiters.shift();
|
|
102
|
+
if (next) next(msg);
|
|
103
|
+
}
|
|
104
|
+
});
|
|
105
|
+
});
|
|
106
|
+
return session;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
return {
|
|
110
|
+
async ask(question, timeoutMs = 3000) {
|
|
111
|
+
let live;
|
|
112
|
+
try {
|
|
113
|
+
live = await open();
|
|
114
|
+
} catch {
|
|
115
|
+
return null;
|
|
116
|
+
}
|
|
117
|
+
if (!live?.ok) return null;
|
|
118
|
+
return new Promise((resolve) => {
|
|
119
|
+
// A timed-out waiter is retired, not merely resolved. Leaving it queued
|
|
120
|
+
// sent the next answer to it instead of to the next asker, and every
|
|
121
|
+
// call after that was off by one.
|
|
122
|
+
let done = false;
|
|
123
|
+
const waiter = (msg) => {
|
|
124
|
+
if (done) return;
|
|
125
|
+
done = true;
|
|
126
|
+
clearTimeout(timer);
|
|
127
|
+
resolve(msg);
|
|
128
|
+
};
|
|
129
|
+
const timer = setTimeout(() => {
|
|
130
|
+
if (done) return;
|
|
131
|
+
done = true;
|
|
132
|
+
const i = live.waiters.indexOf(waiter);
|
|
133
|
+
if (i >= 0) live.waiters.splice(i, 1);
|
|
134
|
+
resolve(null);
|
|
135
|
+
}, timeoutMs);
|
|
136
|
+
live.waiters.push(waiter);
|
|
137
|
+
try {
|
|
138
|
+
live.child.stdin.write(`${JSON.stringify(question)}\n`);
|
|
139
|
+
} catch {
|
|
140
|
+
clearTimeout(timer);
|
|
141
|
+
done = true;
|
|
142
|
+
resolve(null);
|
|
143
|
+
}
|
|
144
|
+
});
|
|
145
|
+
},
|
|
146
|
+
async status() {
|
|
147
|
+
const live = await open();
|
|
148
|
+
return live?.ok ? { ok: true } : { ok: false, reason: live?.reason ?? 'unavailable' };
|
|
149
|
+
},
|
|
150
|
+
close() {
|
|
151
|
+
try { session?.child?.kill(); } catch { /* already gone */ }
|
|
152
|
+
session = null;
|
|
153
|
+
},
|
|
154
|
+
};
|
|
155
|
+
}
|
package/src/matching.js
CHANGED
|
@@ -337,9 +337,67 @@ function sameControl(a, b) {
|
|
|
337
337
|
// and a map built before that still resolves.
|
|
338
338
|
if (isAxTarget(a) && !isAxTarget(b) && sameElementSeenTwice(a, b)) return true;
|
|
339
339
|
if (isAxTarget(b) && !isAxTarget(a) && sameElementSeenTwice(b, a)) return true;
|
|
340
|
+
// And a caption sitting on the control it names. Containment above requires
|
|
341
|
+
// the container to be a recognised hit target, and a React Native composite
|
|
342
|
+
// is a generic element — so a form row published its caption and its select
|
|
343
|
+
// with the same label, 152 points apart, and nothing merged them. They are
|
|
344
|
+
// one control seen twice, not two candidates: answering `ambiguous` here
|
|
345
|
+
// costs a round trip to choose between a thing and its own name.
|
|
346
|
+
if (isCaptionFor(a, b) || isCaptionFor(b, a)) return true;
|
|
340
347
|
return false;
|
|
341
348
|
}
|
|
342
349
|
|
|
350
|
+
/** Does `caption` merely name `control`, overlapping it, with the same label? */
|
|
351
|
+
function isCaptionFor(caption, control) {
|
|
352
|
+
if (!namesOnly(caption) || namesOnly(control)) return false;
|
|
353
|
+
if (norm(caption.label) !== norm(control.label)) return false;
|
|
354
|
+
return overlaps(caption.frame, control.frame);
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/** Any shared area at all — a caption's box often clears its control's by a point or two. */
|
|
358
|
+
function overlaps(a, b) {
|
|
359
|
+
if (!a || !b) return false;
|
|
360
|
+
return a.x < b.x + b.width && b.x < a.x + a.width
|
|
361
|
+
&& a.y < b.y + b.height && b.y < a.y + a.height;
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
/**
|
|
365
|
+
* Roles that only ever *name* a control, never are one.
|
|
366
|
+
*
|
|
367
|
+
* The tree publishes a caption and the control it captions with the same label,
|
|
368
|
+
* and a React Native composite (a `native-base` Select, say) surfaces as a
|
|
369
|
+
* generic element that `INTERACTIVE_ROLE` does not recognise. So `tap "Problem"`
|
|
370
|
+
* resolved to the caption at (49,486) and did nothing, while the select sat at
|
|
371
|
+
* (201,496) — reported as half of the single largest cost in a real-app run,
|
|
372
|
+
* because it means neither selector can be trusted.
|
|
373
|
+
*
|
|
374
|
+
* `collapseSamePlace` already prefers a hit target over the text printed on it,
|
|
375
|
+
* but only once `sameControl` has decided they are the same place. These two
|
|
376
|
+
* were 152 points apart with a generic role, so nothing merged them.
|
|
377
|
+
*/
|
|
378
|
+
const NAMING_ONLY_ROLE = /^(statictext|text|label|heading|image)$/i;
|
|
379
|
+
|
|
380
|
+
const namesOnly = (t) => NAMING_ONLY_ROLE.test(t?.type ?? '');
|
|
381
|
+
|
|
382
|
+
/**
|
|
383
|
+
* A caption never wins over a control that answers the same name.
|
|
384
|
+
*
|
|
385
|
+
* Applied only when the two are close enough in score to be answering the same
|
|
386
|
+
* question — a heading is still reachable when nothing else matches, which is
|
|
387
|
+
* why this promotes rather than filters. Anything further apart than
|
|
388
|
+
* `CAPTION_MARGIN` is a different match, not the same match seen twice.
|
|
389
|
+
*/
|
|
390
|
+
export const CAPTION_MARGIN = 0.2;
|
|
391
|
+
|
|
392
|
+
export function preferTheControl(ranked) {
|
|
393
|
+
if (!ranked.length || !namesOnly(ranked[0].target)) return ranked;
|
|
394
|
+
const lead = ranked[0].score;
|
|
395
|
+
const i = ranked.findIndex((c, idx) => idx > 0 && !namesOnly(c.target) && lead - c.score <= CAPTION_MARGIN);
|
|
396
|
+
if (i < 0) return ranked;
|
|
397
|
+
const promoted = { ...ranked[i], reasons: [...ranked[i].reasons, 'the control, not the caption naming it'] };
|
|
398
|
+
return [promoted, ...ranked.filter((_, idx) => idx !== i)];
|
|
399
|
+
}
|
|
400
|
+
|
|
343
401
|
function collapseSamePlace(ranked) {
|
|
344
402
|
const kept = [];
|
|
345
403
|
for (const c of ranked) {
|
|
@@ -351,6 +409,11 @@ function collapseSamePlace(ranked) {
|
|
|
351
409
|
// Prefer the real hit target: an accessibility element over OCR's reading of
|
|
352
410
|
// it, and an interactive role over a caption sitting inside it.
|
|
353
411
|
const better = (candidate, incumbent) => {
|
|
412
|
+
// A caption loses to what it names before anything else is considered:
|
|
413
|
+
// a `StaticText` is never the tap target when the control it labels is
|
|
414
|
+
// right there, whatever either one's source.
|
|
415
|
+
if (namesOnly(incumbent.target) && !namesOnly(candidate.target)) return true;
|
|
416
|
+
if (namesOnly(candidate.target) && !namesOnly(incumbent.target)) return false;
|
|
354
417
|
if (isAxTarget(candidate.target) && !isAxTarget(incumbent.target)) return true;
|
|
355
418
|
if (!isAxTarget(candidate.target) && isAxTarget(incumbent.target)) return false;
|
|
356
419
|
return INTERACTIVE_ROLE.test(candidate.target.type ?? '')
|
|
@@ -368,7 +431,7 @@ function collapseSamePlace(ranked) {
|
|
|
368
431
|
* @returns {{status: 'ok'|'ambiguous'|'none', target?, score?, reasons?, alternatives?}}
|
|
369
432
|
*/
|
|
370
433
|
export function resolve(targets, intent, options = {}) {
|
|
371
|
-
const ranked = collapseSamePlace(rank(targets, intent, options).filter((c) => c.score >= MINIMUM_SCORE));
|
|
434
|
+
const ranked = preferTheControl(collapseSamePlace(rank(targets, intent, options).filter((c) => c.score >= MINIMUM_SCORE)));
|
|
372
435
|
if (!ranked.length) return { status: 'none', alternatives: [] };
|
|
373
436
|
const [best, second] = ranked;
|
|
374
437
|
if (second && best.score - second.score < AMBIGUITY_MARGIN) {
|