simframe 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -5
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +143 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +281 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1825 -38
- package/src/analyze.js +70 -0
- package/src/cli.js +214 -15
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +193 -11
- package/src/index.js +428 -16
- package/src/input.js +115 -8
- package/src/localhelper.js +155 -0
- package/src/matching.js +119 -3
- package/src/mcp.js +319 -27
- package/src/metrics.js +134 -8
- package/src/navigate.js +10 -7
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +3 -2
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +109 -10
- package/src/supervisor.js +117 -0
- package/src/view.js +396 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/src/metrics.js
CHANGED
|
@@ -40,8 +40,16 @@ export const FACULTY = {
|
|
|
40
40
|
no_plan: 'exploration (Phase 14)',
|
|
41
41
|
};
|
|
42
42
|
|
|
43
|
-
/**
|
|
44
|
-
|
|
43
|
+
/**
|
|
44
|
+
* Faculties that exist.
|
|
45
|
+
*
|
|
46
|
+
* Phase 11 landed the first one, so `verification_failed` no longer maps to
|
|
47
|
+
* something unbuilt — which changes what those records *mean*. Before, they
|
|
48
|
+
* were a queue waiting on a phase. Now they are evidence that the phase which
|
|
49
|
+
* shipped is not sufficient, and that is a more useful thing for the log to be
|
|
50
|
+
* able to say than a count of things nobody has written yet.
|
|
51
|
+
*/
|
|
52
|
+
export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
|
|
45
53
|
|
|
46
54
|
function metricPaths(udid) {
|
|
47
55
|
const dir = store.deviceDir(udid);
|
|
@@ -111,9 +119,27 @@ export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts
|
|
|
111
119
|
* matching error strings at the boundary — a regexed message is a reason that
|
|
112
120
|
* silently becomes "unknown" the day somebody rewords it.
|
|
113
121
|
*/
|
|
114
|
-
export function tag(err, reason, { candidates = [], tried = [] } = {}) {
|
|
122
|
+
export function tag(err, reason, { candidates = [], tried = [], ambiguous = false, intent = null } = {}) {
|
|
115
123
|
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
116
|
-
|
|
124
|
+
// `ambiguous` is narrower than the reason, and that is the point. Two very
|
|
125
|
+
// different failures both tag `ambiguous_intent`: the target is on screen
|
|
126
|
+
// several times over, and the target is not on screen at all on a screen we
|
|
127
|
+
// thought we knew. Only the first is resolvable by *choosing*, and only the
|
|
128
|
+
// first tells a waiting caller that waiting is pointless — the thing it is
|
|
129
|
+
// waiting for has already arrived.
|
|
130
|
+
// `intent` is the goal in the caller's own words, recorded as a field rather
|
|
131
|
+
// than left in the prose of `detail`.
|
|
132
|
+
//
|
|
133
|
+
// Phase 17's go/no-go asks whether an on-device model would pick the element
|
|
134
|
+
// Claude picked, given the goal and the element list. The element list is
|
|
135
|
+
// here as `candidates` and the eventual choice is recoverable from the
|
|
136
|
+
// graph — the tap that finally worked on this screen becomes a verified edge
|
|
137
|
+
// carrying its own step. The goal was the missing third, and it was sitting
|
|
138
|
+
// inside a sentence: `"X" matches 3 things on this screen — say which…`.
|
|
139
|
+
// Regexing it back out at export time is the exact habit this file exists to
|
|
140
|
+
// avoid, and it would silently return nothing the day that sentence is
|
|
141
|
+
// reworded.
|
|
142
|
+
err.escalation = { reason, candidates, tried, ambiguous, intent };
|
|
117
143
|
return err;
|
|
118
144
|
}
|
|
119
145
|
|
|
@@ -213,8 +239,64 @@ export function fingerprintNow(udid, screenmap) {
|
|
|
213
239
|
* measurement's clothes, and every number in this project is supposed to say
|
|
214
240
|
* where it came from.
|
|
215
241
|
*/
|
|
242
|
+
/**
|
|
243
|
+
* Which run of which program wrote a record.
|
|
244
|
+
*
|
|
245
|
+
* The log is per-device and, until now, anonymous — so two agents driving one
|
|
246
|
+
* booted simulator wrote one interleaved file with no way to separate them.
|
|
247
|
+
* Measured on the bench device in a single evening: 57 records to 81, a third
|
|
248
|
+
* of the new ones naming screens from an app the suite has never launched.
|
|
249
|
+
*
|
|
250
|
+
* That is not a corrupted file, it is a corrupted instrument. CLAUDE.md makes
|
|
251
|
+
* the reason breakdown of this log the thing that chooses which faculty gets
|
|
252
|
+
* built next, and a breakdown that silently pools two sessions errs toward
|
|
253
|
+
* whichever of them made more mistakes — which is not the same question as
|
|
254
|
+
* which faculty is missing.
|
|
255
|
+
*
|
|
256
|
+
* A pid alone would not do: pids are reused, and the useful grouping is "one
|
|
257
|
+
* agent's run", which for the MCP server is the life of the process and for
|
|
258
|
+
* the CLI is a single command. So: the start time, the pid, and a random tail,
|
|
259
|
+
* computed once per process. `client` says what kind of process it was, since
|
|
260
|
+
* "the MCP server did this" and "somebody ran a CLI command" deserve different
|
|
261
|
+
* readings of the same reason.
|
|
262
|
+
*
|
|
263
|
+
* Deliberately not a device id, a username, or anything about the machine. This
|
|
264
|
+
* file is committed to a public repo in summary form, and the question it has
|
|
265
|
+
* to answer is "was this all one agent", which needs no identity to answer.
|
|
266
|
+
*/
|
|
267
|
+
const SESSION_ID = process.env.SIMFRAME_SESSION
|
|
268
|
+
? String(process.env.SIMFRAME_SESSION).slice(0, 64)
|
|
269
|
+
: `${process.pid.toString(36)}-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 8)}`;
|
|
270
|
+
|
|
271
|
+
/*
|
|
272
|
+
* Minting the id from the pid was right for the MCP server, which is one
|
|
273
|
+
* long-lived process, and wrong for everything else. A CLI-driven agent starts
|
|
274
|
+
* a process per command, so it got one "session" per command: on the benchmark
|
|
275
|
+
* device, 33 session ids for 46 records, 30 of them holding a single record.
|
|
276
|
+
* `escalations --session` was therefore unable to answer the one question it
|
|
277
|
+
* exists for, and Phase 17's go/no-go step 1 — "filter to one session id" —
|
|
278
|
+
* had nothing to filter. `SIMFRAME_SESSION` lets a caller that knows it is one
|
|
279
|
+
* session say so; the per-process id stays the default.
|
|
280
|
+
*/
|
|
281
|
+
|
|
282
|
+
/** How this process is being used, for reading a breakdown afterwards. */
|
|
283
|
+
function clientKind() {
|
|
284
|
+
const argv = process.argv.join(' ');
|
|
285
|
+
if (/\bmcp\b/.test(argv)) return 'mcp';
|
|
286
|
+
if (/bench-hpi|scripts\//.test(argv)) return 'script';
|
|
287
|
+
if (/cli\.js|\bsimframe\b/.test(argv)) return 'cli';
|
|
288
|
+
return 'library';
|
|
289
|
+
}
|
|
290
|
+
const CLIENT = clientKind();
|
|
291
|
+
|
|
292
|
+
/** The session this process's records belong to. Exported for `escalations`. */
|
|
293
|
+
export const sessionId = () => SESSION_ID;
|
|
294
|
+
export const clientName = () => CLIENT;
|
|
295
|
+
|
|
216
296
|
export function recordEscalation(udid, {
|
|
217
297
|
flowId = null,
|
|
298
|
+
flowName = null,
|
|
299
|
+
intent = null,
|
|
218
300
|
stepIndex = null,
|
|
219
301
|
fingerprint = null,
|
|
220
302
|
reason,
|
|
@@ -229,7 +311,16 @@ export function recordEscalation(udid, {
|
|
|
229
311
|
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
230
312
|
const record = {
|
|
231
313
|
timestamp: new Date().toISOString(),
|
|
314
|
+
// Added after the log turned out to pool two agents' work invisibly. Both
|
|
315
|
+
// are cheap and neither is derivable afterwards, which is the test for
|
|
316
|
+
// whether a field belongs in a log at all.
|
|
317
|
+
session_id: SESSION_ID,
|
|
318
|
+
client: CLIENT,
|
|
232
319
|
flow_id: flowId,
|
|
320
|
+
flow_name: flowName,
|
|
321
|
+
// What was asked for, in the caller's words. Ground truth for Phase 17's
|
|
322
|
+
// go/no-go, and on its own it answers "what kind of decision is costing us".
|
|
323
|
+
intent: intent ? String(intent).slice(0, 120) : null,
|
|
233
324
|
step_index: stepIndex,
|
|
234
325
|
screen_fingerprint: fingerprint,
|
|
235
326
|
reason,
|
|
@@ -408,14 +499,31 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
408
499
|
* rate is 1.0 by construction and says nothing. The per-reason breakdown is
|
|
409
500
|
* the part that decides the next phase, and it is informative today.
|
|
410
501
|
*/
|
|
411
|
-
export function breakdown(records) {
|
|
502
|
+
export function breakdown(records, { session = null, flow = null } = {}) {
|
|
412
503
|
const byReason = {};
|
|
413
504
|
for (const r of REASONS) byReason[r] = 0;
|
|
414
505
|
const byScreen = new Map();
|
|
415
506
|
const byOutcome = {};
|
|
507
|
+
const bySession = new Map();
|
|
508
|
+
const byFlow = new Map();
|
|
416
509
|
let avoidable = 0;
|
|
417
|
-
|
|
510
|
+
let unattributed = 0;
|
|
511
|
+
// Filtering happens here rather than at the call site so `total` and every
|
|
512
|
+
// rate below it describe the same set of records.
|
|
513
|
+
const kept = records.filter((r) => (session ? r?.session_id === session : true))
|
|
514
|
+
.filter((r) => (flow ? r?.flow_name === flow : true));
|
|
515
|
+
for (const r of kept) {
|
|
418
516
|
if (!REASONS.includes(r?.reason)) continue;
|
|
517
|
+
// Records written before sessions were recorded cannot be attributed, and
|
|
518
|
+
// saying how many there are is the difference between a breakdown that
|
|
519
|
+
// pools two agents and one that says it might be.
|
|
520
|
+
if (r.session_id) {
|
|
521
|
+
const key = `${r.session_id}|${r.client ?? '?'}`;
|
|
522
|
+
bySession.set(key, (bySession.get(key) ?? 0) + 1);
|
|
523
|
+
} else {
|
|
524
|
+
unattributed += 1;
|
|
525
|
+
}
|
|
526
|
+
if (r.flow_name) byFlow.set(r.flow_name, (byFlow.get(r.flow_name) ?? 0) + 1);
|
|
419
527
|
byReason[r.reason] += 1;
|
|
420
528
|
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
421
529
|
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
@@ -423,9 +531,25 @@ export function breakdown(records) {
|
|
|
423
531
|
const key = r.screen_fingerprint ?? '(no fingerprint)';
|
|
424
532
|
byScreen.set(key, (byScreen.get(key) ?? 0) + 1);
|
|
425
533
|
}
|
|
426
|
-
const total =
|
|
534
|
+
const total = kept.filter((r) => REASONS.includes(r?.reason)).length;
|
|
535
|
+
const sessions = [...bySession.entries()]
|
|
536
|
+
.map(([key, count]) => {
|
|
537
|
+
const [id, client] = key.split('|');
|
|
538
|
+
return { session_id: id, client, count };
|
|
539
|
+
})
|
|
540
|
+
.sort((a, b) => b.count - a.count);
|
|
427
541
|
return {
|
|
428
542
|
total,
|
|
543
|
+
// The log is per-device and shared: two agents on one booted simulator
|
|
544
|
+
// write one interleaved file. More than one session here means the counts
|
|
545
|
+
// below are a pool, and CLAUDE.md uses those counts to choose a phase.
|
|
546
|
+
sessions,
|
|
547
|
+
session_count: sessions.length,
|
|
548
|
+
unattributed,
|
|
549
|
+
// Any unattributed record at all makes this a pool: the whole point is
|
|
550
|
+
// that they cannot be told apart, and 92 of them is not "one session".
|
|
551
|
+
pooled: sessions.length > 1 || unattributed > 0,
|
|
552
|
+
by_flow: Object.fromEntries([...byFlow.entries()].sort((a, b) => b[1] - a[1])),
|
|
429
553
|
by_reason: byReason,
|
|
430
554
|
by_outcome: byOutcome,
|
|
431
555
|
faculty: Object.fromEntries(
|
|
@@ -437,7 +561,9 @@ export function breakdown(records) {
|
|
|
437
561
|
.sort((a, b) => b[1] - a[1])
|
|
438
562
|
.slice(0, 10)
|
|
439
563
|
.map(([fingerprint, count]) => ({ fingerprint, count })),
|
|
440
|
-
|
|
564
|
+
// `kept`, not `records` — a filtered breakdown that reports the whole
|
|
565
|
+
// log's model turns is the same class of mistake as pooling two sessions.
|
|
566
|
+
model_turns_spent: kept.reduce((acc, r) => acc + (r.model_turns_spent ?? 0), 0),
|
|
441
567
|
};
|
|
442
568
|
}
|
|
443
569
|
|
package/src/navigate.js
CHANGED
|
@@ -41,12 +41,15 @@ export function stepFor(edge) {
|
|
|
41
41
|
* count. The reason mapping lives in metrics.PLAN_REASONS so the five reasons
|
|
42
42
|
* have one owner.
|
|
43
43
|
*/
|
|
44
|
-
function refuse(udid, result, { detail = null } = {}) {
|
|
44
|
+
function refuse(udid, result, { detail = null, flowName = null } = {}) {
|
|
45
45
|
try {
|
|
46
46
|
const reason = metrics.PLAN_REASONS[result.reason];
|
|
47
47
|
if (reason) {
|
|
48
48
|
metrics.recordEscalation(udid, {
|
|
49
49
|
reason,
|
|
50
|
+
// A refusal by `goto` is about a destination and one by `flow run` is
|
|
51
|
+
// about a named flow. Either is what a breakdown wants to group by.
|
|
52
|
+
flowName,
|
|
50
53
|
fingerprint: metrics.fingerprintNow(udid, screenmap),
|
|
51
54
|
outcome: 'escalated_to_model',
|
|
52
55
|
detail: detail ?? result.reason,
|
|
@@ -70,8 +73,8 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
70
73
|
const udid = device.udid;
|
|
71
74
|
|
|
72
75
|
const found = graph.findScreen(udid, target);
|
|
73
|
-
if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"` });
|
|
74
|
-
if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens` });
|
|
76
|
+
if (!found) return refuse(udid, { ok: false, reason: 'unknown-screen', known: knownScreens(udid) }, { detail: `no screen matches "${target}"`, flowName: `goto:${target}` });
|
|
77
|
+
if (found.ambiguous) return refuse(udid, { ok: false, reason: 'ambiguous', candidates: found.ambiguous }, { detail: `"${target}" fits ${found.ambiguous.length} screens`, flowName: `goto:${target}` });
|
|
75
78
|
|
|
76
79
|
const here = await api.screenIdentity(udid, {});
|
|
77
80
|
if (here.hash === found.node.hash) {
|
|
@@ -84,12 +87,12 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
84
87
|
// `Cannot read properties of null (reading 'slice')` instead of answering.
|
|
85
88
|
// Not hypothetical on Android, where README's own table puts the launcher at
|
|
86
89
|
// one token.
|
|
87
|
-
if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name });
|
|
90
|
+
if (!here.hash) return refuse(udid, { ok: false, reason: 'no-identity', to: found.name }, { flowName: `goto:${target}` });
|
|
88
91
|
const path_ = graph.route(udid, { hash: here.hash, tokens: here.tokens }, found.node.hash);
|
|
89
|
-
if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name });
|
|
92
|
+
if (!path_) return refuse(udid, { ok: false, reason: 'no-route', from: here.hash.slice(0, 8), to: found.name }, { flowName: `goto:${target}` });
|
|
90
93
|
|
|
91
94
|
const steps = path_.map(stepFor);
|
|
92
|
-
if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name });
|
|
95
|
+
if (steps.some((s) => !s)) return refuse(udid, { ok: false, reason: 'unreplayable-edge', to: found.name }, { flowName: `goto:${target}` });
|
|
93
96
|
|
|
94
97
|
const result = await runScript(udid, { steps, stopOnUnexpected: true, ...runOptions });
|
|
95
98
|
const arrived = await api.screenIdentity(udid, {});
|
|
@@ -148,7 +151,7 @@ export function listFlows(udid) {
|
|
|
148
151
|
export async function runFlow(deviceQuery, name, { options, ...runOptions } = {}) {
|
|
149
152
|
const { device } = await api.ensureDaemon(deviceQuery, options);
|
|
150
153
|
const flow = loadFlow(device.udid, name);
|
|
151
|
-
if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"
|
|
154
|
+
if (!flow) return refuse(device.udid, { ok: false, reason: 'unknown-flow', known: listFlows(device.udid).map((f) => f.name) }, { detail: `no saved flow "${name}"`, flowName: name });
|
|
152
155
|
// A replayed flow knows its own name, so its record can be compared against
|
|
153
156
|
// a human doing the same thing. `minSteps` comes from the flow definition or
|
|
154
157
|
// stays null — the step count of a recorded route is not a claim about the
|
package/src/ocr.js
CHANGED
|
@@ -19,6 +19,17 @@ const BIN = path.join(BIN_DIR, 'ocr');
|
|
|
19
19
|
|
|
20
20
|
let ready = null;
|
|
21
21
|
|
|
22
|
+
/**
|
|
23
|
+
* Which recognition level to ask Vision for.
|
|
24
|
+
*
|
|
25
|
+
* `accurate` is the default and what CLAUDE.md fixes; `fast` is the other thing
|
|
26
|
+
* Vision offers. Exposed so the pair can be scored against each other instead
|
|
27
|
+
* of one of them being a constant nobody measured.
|
|
28
|
+
*/
|
|
29
|
+
export function level() {
|
|
30
|
+
return String(process.env.SIMFRAME_OCR ?? '').toLowerCase() === 'fast' ? 'fast' : 'accurate';
|
|
31
|
+
}
|
|
32
|
+
|
|
22
33
|
/** Compile once, then reuse. Recompiles only if the source is newer than the binary. */
|
|
23
34
|
export async function ensureBinary() {
|
|
24
35
|
if (ready) return ready;
|
|
@@ -55,7 +66,13 @@ export async function ensureBinary() {
|
|
|
55
66
|
export async function readText(pngFile, { density = 3 } = {}) {
|
|
56
67
|
const built = await ensureBinary();
|
|
57
68
|
if (!built.available) throw new Error(built.reason);
|
|
58
|
-
|
|
69
|
+
// The recognition level rides in the environment rather than in argv, so the
|
|
70
|
+
// Swift side keeps its one-argument contract and an older binary still works.
|
|
71
|
+
const { stdout } = await run(built.binary, [pngFile], {
|
|
72
|
+
timeout: 30_000,
|
|
73
|
+
maxBuffer: 16 << 20,
|
|
74
|
+
env: { ...process.env, SIMFRAME_OCR: level() },
|
|
75
|
+
});
|
|
59
76
|
const raw = JSON.parse(stdout || '[]');
|
|
60
77
|
return raw.map((r) => ({
|
|
61
78
|
text: r.text,
|
package/src/planner.js
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The local planner tier: a ranker, behind a flag, that may only reorder.
|
|
3
|
+
*
|
|
4
|
+
* **Why this exists at all, given Phase 17 was a no-go.** That phase asked a
|
|
5
|
+
* local model to *choose the next element*, and the answer was that the matcher
|
|
6
|
+
* already does — 37 of 40 real decisions. This is the complement and the one
|
|
7
|
+
* case a string matcher structurally cannot do: the goal matches **nothing** on
|
|
8
|
+
* screen, and something has to guess which container leads to it. "Change my
|
|
9
|
+
* username" shares no prefix, synonym or typo distance with "Account".
|
|
10
|
+
*
|
|
11
|
+
* Measured on this machine, six hand-written cases: 5 of 6 top-1, 6 of 6 top-3,
|
|
12
|
+
* median 564 ms warm. See `docs/BENCHMARKS.md`. Against a model round trip at
|
|
13
|
+
* 10–16 s that is roughly twenty times cheaper; against the honest baseline —
|
|
14
|
+
* breadth-first ordering, which needs no model — it won five of six.
|
|
15
|
+
*
|
|
16
|
+
* **What it is allowed to do, and it is deliberately almost nothing.** It
|
|
17
|
+
* reorders a list of candidates the caller has already permitted and will try
|
|
18
|
+
* in some order regardless. It cannot invent a label, cannot choose an action,
|
|
19
|
+
* cannot see pixels, and never runs on a destructive label because the caller
|
|
20
|
+
* filtered those out before asking (`src/vocabulary.js`). If it is wrong the
|
|
21
|
+
* exploration budget simply tries the next one. That is strictly weaker
|
|
22
|
+
* authority than Phase 17 proposed, which is what makes it safe to try.
|
|
23
|
+
*
|
|
24
|
+
* **Off unless asked.** `SIMFRAME_PLANNER=apple` turns it on; anything else,
|
|
25
|
+
* or any failure at all, degrades to `null` and the caller keeps its own order.
|
|
26
|
+
* `doctor` reports which. CI runs with it off.
|
|
27
|
+
*/
|
|
28
|
+
import { spawn, execFile } from 'node:child_process';
|
|
29
|
+
import fs from 'node:fs';
|
|
30
|
+
import path from 'node:path';
|
|
31
|
+
import { fileURLToPath } from 'node:url';
|
|
32
|
+
import { promisify } from 'node:util';
|
|
33
|
+
import * as store from './store.js';
|
|
34
|
+
|
|
35
|
+
const run = promisify(execFile);
|
|
36
|
+
const SOURCE = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'native', 'rank.swift');
|
|
37
|
+
const BIN = path.join(store.ROOT, 'bin', 'rank');
|
|
38
|
+
|
|
39
|
+
/** Which backend the caller asked for. Absent means no local planner. */
|
|
40
|
+
export function requested(options) {
|
|
41
|
+
// Per call first, then the environment, for the reason in `sensorMode`: an
|
|
42
|
+
// MCP server's environment is fixed when it spawns, so a tester could not
|
|
43
|
+
// switch backends inside one session and a round came back with one arm of
|
|
44
|
+
// its A/B unrun.
|
|
45
|
+
const raw = String(options?.planner ?? process.env.SIMFRAME_PLANNER ?? '').trim().toLowerCase();
|
|
46
|
+
if (!raw || raw === 'none' || raw === 'off' || raw === '0' || raw === 'false') return null;
|
|
47
|
+
return raw;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
let building = null;
|
|
51
|
+
|
|
52
|
+
export async function ensureBinary() {
|
|
53
|
+
if (building) return building;
|
|
54
|
+
building = (async () => {
|
|
55
|
+
try {
|
|
56
|
+
const src = fs.statSync(SOURCE).mtimeMs;
|
|
57
|
+
const bin = fs.existsSync(BIN) ? fs.statSync(BIN).mtimeMs : 0;
|
|
58
|
+
if (bin > src) return { available: true, binary: BIN };
|
|
59
|
+
} catch {
|
|
60
|
+
return { available: false, reason: 'the ranker source is missing from this install' };
|
|
61
|
+
}
|
|
62
|
+
try {
|
|
63
|
+
fs.mkdirSync(path.dirname(BIN), { recursive: true });
|
|
64
|
+
// `swiftc`, not `xcrun swiftc`: nothing above the platform boundary may
|
|
65
|
+
// name a platform tool, and the boundary test catches it. `src/ocr.js`
|
|
66
|
+
// set this precedent — a compiler is not a device tool.
|
|
67
|
+
await run('swiftc', ['-O', SOURCE, '-o', BIN], { timeout: 180_000 });
|
|
68
|
+
return { available: true, binary: BIN };
|
|
69
|
+
} catch (err) {
|
|
70
|
+
building = null; // let a later call retry once a toolchain is present
|
|
71
|
+
return {
|
|
72
|
+
available: false,
|
|
73
|
+
reason: err.code === 'ENOENT'
|
|
74
|
+
? 'swiftc is not installed, so the local planner cannot be built (install Xcode command line tools)'
|
|
75
|
+
: `could not build the local planner: ${String(err.message).split('\n')[0]}`,
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
})();
|
|
79
|
+
return building;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
let session = null;
|
|
83
|
+
|
|
84
|
+
/** Start the helper once and keep it, because the first answer pays model load. */
|
|
85
|
+
async function open() {
|
|
86
|
+
if (session) return session;
|
|
87
|
+
const built = await ensureBinary();
|
|
88
|
+
if (!built.available) return { ok: false, reason: built.reason };
|
|
89
|
+
session = await new Promise((resolve) => {
|
|
90
|
+
const child = spawn(built.binary, [], { stdio: ['pipe', 'pipe', 'ignore'] });
|
|
91
|
+
// Deliberately NOT unref'd. Unreffing the child's stdout unreferences the
|
|
92
|
+
// very pipe every request waits on, so the process exited silently in the
|
|
93
|
+
// middle of an await — a flow that printed nothing and returned 0. The
|
|
94
|
+
// helper is closed explicitly instead, by whoever opened it.
|
|
95
|
+
let buffer = '';
|
|
96
|
+
const waiters = [];
|
|
97
|
+
let settled = false;
|
|
98
|
+
const fail = (reason) => {
|
|
99
|
+
if (!settled) { settled = true; resolve({ ok: false, reason }); }
|
|
100
|
+
while (waiters.length) waiters.shift()(null);
|
|
101
|
+
};
|
|
102
|
+
child.on('error', (err) => fail(`the local planner would not start: ${err.message}`));
|
|
103
|
+
child.on('exit', () => { session = null; fail('the local planner exited'); });
|
|
104
|
+
child.stdout.on('data', (chunk) => {
|
|
105
|
+
buffer += chunk;
|
|
106
|
+
let i = buffer.indexOf('\n');
|
|
107
|
+
while (i >= 0) {
|
|
108
|
+
const line = buffer.slice(0, i).trim();
|
|
109
|
+
buffer = buffer.slice(i + 1);
|
|
110
|
+
i = buffer.indexOf('\n');
|
|
111
|
+
if (!line) continue;
|
|
112
|
+
let msg;
|
|
113
|
+
try { msg = JSON.parse(line); } catch { continue; }
|
|
114
|
+
if (!settled) {
|
|
115
|
+
settled = true;
|
|
116
|
+
if (msg.ready) resolve({ ok: true, child, waiters });
|
|
117
|
+
else resolve({ ok: false, reason: msg.unavailable ?? 'the local planner did not become ready' });
|
|
118
|
+
continue;
|
|
119
|
+
}
|
|
120
|
+
const next = waiters.shift();
|
|
121
|
+
if (next) next(msg);
|
|
122
|
+
}
|
|
123
|
+
});
|
|
124
|
+
});
|
|
125
|
+
return session;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Reorder `options` by which is likeliest to lead to `goal`.
|
|
130
|
+
*
|
|
131
|
+
* @returns {Promise<string[]|null>} the caller's own order is correct when this
|
|
132
|
+
* is null, which is every failure mode: flag off, no model, a timeout, a
|
|
133
|
+
* parse problem, a paraphrasing answer. Never throws.
|
|
134
|
+
*/
|
|
135
|
+
export async function rank(goal, options, { timeoutMs = 3000, deviceOptions } = {}) {
|
|
136
|
+
if (!requested(deviceOptions)) return null;
|
|
137
|
+
if (!goal || !Array.isArray(options) || options.length < 2) return null;
|
|
138
|
+
let live;
|
|
139
|
+
try {
|
|
140
|
+
live = await open();
|
|
141
|
+
} catch {
|
|
142
|
+
return null;
|
|
143
|
+
}
|
|
144
|
+
if (!live?.ok) return null;
|
|
145
|
+
const answer = await new Promise((resolve) => {
|
|
146
|
+
// A timed-out waiter has to be *retired*, not merely resolved. Leaving it in
|
|
147
|
+
// the queue meant the next answer went to it instead of to the next asker,
|
|
148
|
+
// and every call after that was off by one — which showed up as an
|
|
149
|
+
// exploration run that never finished rather than as an error.
|
|
150
|
+
let done = false;
|
|
151
|
+
const waiter = (msg) => {
|
|
152
|
+
if (done) return;
|
|
153
|
+
done = true;
|
|
154
|
+
clearTimeout(timer);
|
|
155
|
+
resolve(msg);
|
|
156
|
+
};
|
|
157
|
+
const timer = setTimeout(() => {
|
|
158
|
+
if (done) return;
|
|
159
|
+
done = true;
|
|
160
|
+
const i = live.waiters.indexOf(waiter);
|
|
161
|
+
if (i >= 0) live.waiters.splice(i, 1);
|
|
162
|
+
resolve(null);
|
|
163
|
+
}, timeoutMs);
|
|
164
|
+
live.waiters.push(waiter);
|
|
165
|
+
try {
|
|
166
|
+
live.child.stdin.write(`${JSON.stringify({ goal: String(goal), options })}\n`);
|
|
167
|
+
} catch {
|
|
168
|
+
clearTimeout(timer);
|
|
169
|
+
resolve(null);
|
|
170
|
+
}
|
|
171
|
+
});
|
|
172
|
+
if (!answer?.order?.length) return null;
|
|
173
|
+
// It often returns a subset, so its order comes first and ours fills the tail.
|
|
174
|
+
// Trusting it to be exhaustive would silently drop candidates the budget was
|
|
175
|
+
// going to try.
|
|
176
|
+
const ranked = answer.order.filter((label) => options.includes(label));
|
|
177
|
+
const seen = new Set(ranked);
|
|
178
|
+
return [...ranked, ...options.filter((o) => !seen.has(o))];
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/** For `doctor`: what the planner layer is, in one line. */
|
|
182
|
+
export async function status(options) {
|
|
183
|
+
const want = requested(options);
|
|
184
|
+
if (!want) return { planner: 'none', detail: 'not requested (SIMFRAME_PLANNER is unset)' };
|
|
185
|
+
if (want !== 'apple') return { planner: 'none', detail: `no such planner backend: "${want}"` };
|
|
186
|
+
const live = await open();
|
|
187
|
+
if (!live?.ok) return { planner: 'none', detail: live?.reason ?? 'unavailable' };
|
|
188
|
+
return { planner: 'apple', detail: 'Apple Foundation Models, on-device, ranking only' };
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** Let a process exit without waiting on the helper. */
|
|
192
|
+
export function close() {
|
|
193
|
+
try { session?.child?.kill(); } catch { /* already gone */ }
|
|
194
|
+
session = null;
|
|
195
|
+
}
|
package/src/platform/android.js
CHANGED
|
@@ -183,7 +183,8 @@ async function resolveDevice(query, opts) {
|
|
|
183
183
|
new Error(
|
|
184
184
|
`${booted.length} emulators are running and none was named: ` +
|
|
185
185
|
`${booted.map((d) => `${d.name} (${d.udid})`).join(', ')} — name one with --device, ` +
|
|
186
|
-
'or set SIMFRAME_DEVICE to pick a default for this shell'
|
|
186
|
+
'or set SIMFRAME_DEVICE to pick a default for this shell. Over MCP there is no shell: ' +
|
|
187
|
+
'pass "device" once on any call and the rest of the session remembers it',
|
|
187
188
|
),
|
|
188
189
|
{ ambiguous: true },
|
|
189
190
|
);
|
|
@@ -827,7 +828,7 @@ async function launchApp(udid, bundleId, { args = [], env = {}, terminateFirst =
|
|
|
827
828
|
async function terminateApp(udid, bundleId) {
|
|
828
829
|
// `am` reports failure on stdout and still exits 0 — the same trap launchApp
|
|
829
830
|
// and openUrl already check for. Without this, terminating a package that is
|
|
830
|
-
// not installed answered "terminated com.
|
|
831
|
+
// not installed answered "terminated com.example.mistyped".
|
|
831
832
|
const { stdout, stderr } = await adb(udid, ['shell', 'am', 'force-stop', bundleId]);
|
|
832
833
|
const error = /^Error:.*$/m.exec(`${stdout}${stderr}`);
|
|
833
834
|
if (error) {
|
package/src/platform/ios.js
CHANGED
|
@@ -87,7 +87,8 @@ async function resolveDevice(query, opts) {
|
|
|
87
87
|
new Error(
|
|
88
88
|
`${booted.length} simulators are booted and none was named: ` +
|
|
89
89
|
`${booted.map((d) => `${d.name} (${d.udid})`).join(', ')} — name one with --device, ` +
|
|
90
|
-
'or set SIMFRAME_DEVICE to pick a default for this shell'
|
|
90
|
+
'or set SIMFRAME_DEVICE to pick a default for this shell. Over MCP there is no shell: ' +
|
|
91
|
+
'pass "device" once on any call and the rest of the session remembers it',
|
|
91
92
|
),
|
|
92
93
|
{ ambiguous: true },
|
|
93
94
|
);
|
package/src/png.js
CHANGED
|
@@ -178,6 +178,32 @@ export function grayGrid(bmp, cols, rows) {
|
|
|
178
178
|
}
|
|
179
179
|
|
|
180
180
|
/** Nearest-neighbour scale. Only used for contact sheets, where speed beats quality. */
|
|
181
|
+
/**
|
|
182
|
+
* A rectangle out of a bitmap, clamped to it.
|
|
183
|
+
*
|
|
184
|
+
* Exists because a whole screen at 1024px on the long edge cannot answer a
|
|
185
|
+
* question about one control. Reported from a real session: a selected filter
|
|
186
|
+
* chip and an unselected one are indistinguishable at that size, and selection
|
|
187
|
+
* state was the entire question the ticket turned on — so the agent shelled out
|
|
188
|
+
* to `simctl io` and PIL to crop and upscale the chip row, **for every single
|
|
189
|
+
* check**. Their estimate: six round trips.
|
|
190
|
+
*
|
|
191
|
+
* Coordinates are pixels; the caller converts from points, because only the
|
|
192
|
+
* caller knows the density it read them at.
|
|
193
|
+
*/
|
|
194
|
+
export function cropBitmap(bmp, x, y, width, height) {
|
|
195
|
+
const left = Math.max(0, Math.min(bmp.width - 1, Math.round(x)));
|
|
196
|
+
const top = Math.max(0, Math.min(bmp.height - 1, Math.round(y)));
|
|
197
|
+
const w = Math.max(1, Math.min(bmp.width - left, Math.round(width)));
|
|
198
|
+
const h = Math.max(1, Math.min(bmp.height - top, Math.round(height)));
|
|
199
|
+
const out = Buffer.alloc(w * h * 4);
|
|
200
|
+
for (let row = 0; row < h; row += 1) {
|
|
201
|
+
const from = ((top + row) * bmp.width + left) * 4;
|
|
202
|
+
bmp.data.copy(out, row * w * 4, from, from + w * 4);
|
|
203
|
+
}
|
|
204
|
+
return { width: w, height: h, data: out };
|
|
205
|
+
}
|
|
206
|
+
|
|
181
207
|
export function scaleBitmap(bmp, width, height) {
|
|
182
208
|
const out = Buffer.allocUnsafe(width * height * 4);
|
|
183
209
|
for (let y = 0; y < height; y++) {
|
package/src/refs.js
CHANGED
|
@@ -104,17 +104,46 @@ export function parseSelector(query) {
|
|
|
104
104
|
* numbering introduces that labels do not have, and a ref resolved against the
|
|
105
105
|
* wrong screen taps whatever now happens to sit at those coordinates.
|
|
106
106
|
*/
|
|
107
|
-
export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, tolerance = REF_TOLERANCE } = {}) {
|
|
107
|
+
export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, structuralDistance = 0, tolerance = REF_TOLERANCE } = {}) {
|
|
108
108
|
const table = readRefs(udid);
|
|
109
109
|
if (!table) throw new Error(`#${n} means nothing yet — read the screen first (sim_ui, or simframe ui)`);
|
|
110
|
-
|
|
111
|
-
|
|
110
|
+
// The label this number was given to, when there is one. A stale ref is not
|
|
111
|
+
// nothing: the table records what it pointed at, which is enough for the
|
|
112
|
+
// caller to be offered the label instead of a bare refusal.
|
|
113
|
+
const labelFor = table.refs?.find((r) => r.ref === n)?.label ?? null;
|
|
114
|
+
// How long ago these numbers were handed out. Asked for by name: "refs
|
|
115
|
+
// expired (issued 4 calls ago) is actionable in a way this isn't".
|
|
116
|
+
const issued = Number.isFinite(table.at) ? ` refs were numbered ${Math.round((Date.now() - table.at) / 1000)}s ago;` : '';
|
|
117
|
+
// `staleKind` is the difference between "these numbers were drawn on a screen
|
|
118
|
+
// that has since shifted" and "you are somewhere else entirely", and only the
|
|
119
|
+
// first may be recovered by re-resolving the label the number stood for.
|
|
120
|
+
// Both wore the same flag once, and the caller re-resolved across an app
|
|
121
|
+
// switch: `#1` had been "Reminders" in Contacts, matched the status-bar
|
|
122
|
+
// back-to-app breadcrumb "• Reminders" at 0.64, and returned a tappable point
|
|
123
|
+
// in the status bar — a region the map itself refuses to offer. A refusal had
|
|
124
|
+
// become a confident wrong answer.
|
|
125
|
+
const staleError = (why, kind) => Object.assign(
|
|
126
|
+
new Error(`#${n} cannot be trusted here —${issued} ${why}. Read the screen again (sim_ui) to renumber`),
|
|
127
|
+
{ staleRef: true, staleLabel: labelFor, staleKind: kind },
|
|
128
|
+
);
|
|
112
129
|
|
|
113
130
|
// Structural identity first, because it is the question actually being asked:
|
|
114
131
|
// is this the screen those numbers were assigned on? The caller gets it
|
|
115
132
|
// cheaply — screen memory is a file read, not a perception pass.
|
|
116
|
-
|
|
117
|
-
|
|
133
|
+
//
|
|
134
|
+
// But only when the recall that produced it was exact. The identity arrives
|
|
135
|
+
// from `recallNearest`, which matches by layout within a tolerance so that a
|
|
136
|
+
// list with new rows stays one screen; above distance zero it is therefore a
|
|
137
|
+
// guess about *which* remembered screen this is, and a guess cannot be the
|
|
138
|
+
// sole reason to refuse. That mismatch was reported from the field as a
|
|
139
|
+
// refusal on unchanged state — the map had named the screen from the tolerant
|
|
140
|
+
// recall and printed the same header before and after, while this check read
|
|
141
|
+
// the same recall as exact and disagreed with it. Beyond distance zero the
|
|
142
|
+
// pixel backstop below is the one that decides, which is what it is for.
|
|
143
|
+
const exactRecall = structuralDistance === 0 || structuralDistance == null;
|
|
144
|
+
if (exactRecall && table.structuralHash && structuralHash && table.structuralHash !== structuralHash) {
|
|
145
|
+
throw staleError(`this is a different screen (${table.structuralHash.slice(0, 8)}`
|
|
146
|
+
+ ` → ${structuralHash.slice(0, 8)})`, 'identity');
|
|
118
147
|
}
|
|
119
148
|
// Nothing recognises the screen we are on, so nothing can vouch for the
|
|
120
149
|
// numbers. Refusing costs a re-read; guessing taps whatever is at those
|
|
@@ -128,9 +157,23 @@ export function resolveRef(udid, n, { structuralHash, layoutHash, screenKnown, t
|
|
|
128
157
|
// other — measured: refs numbered on the springboard resolved happily on a
|
|
129
158
|
// different screen because both hashes were degenerate. A hash with almost
|
|
130
159
|
// no bits set is not evidence of anything.
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
160
|
+
// Reported three times in one session as `#4 was numbered on a different
|
|
161
|
+
// screen (03003714 → 03003714)` — a message that says the screen changed
|
|
162
|
+
// while showing that it did not, and left the reporter unable to tell a real
|
|
163
|
+
// move from a false positive. The cause was this branch printing eight
|
|
164
|
+
// characters of a **72-character** perceptual hash: its leading characters
|
|
165
|
+
// encode coarse structure, which is the very reason this comparison is a
|
|
166
|
+
// distance against a tolerance rather than an equality, so prefixes coincide
|
|
167
|
+
// routinely while the hashes differ. So this says the distance, and says that
|
|
168
|
+
// it is pixels rather than identity — a different thing from the branch above,
|
|
169
|
+
// which had been wearing the same sentence.
|
|
170
|
+
const drift = layoutHash && table.layoutHash && informative(table.layoutHash) && informative(layoutHash)
|
|
171
|
+
? hashDistance(table.layoutHash, layoutHash)
|
|
172
|
+
: null;
|
|
173
|
+
if (drift != null && drift > tolerance) {
|
|
174
|
+
throw staleError(`the screen has moved too far from where these refs were numbered`
|
|
175
|
+
+ ` (layout distance ${drift}, tolerance ${tolerance}) — the identity may be unchanged;`
|
|
176
|
+
+ ' this is a pixel measurement, not a different screen', 'drift');
|
|
134
177
|
}
|
|
135
178
|
const hit = table.refs.find((r) => r.ref === n);
|
|
136
179
|
if (!hit) {
|