simframe 0.11.0 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +44 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +29 -1
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +20 -2
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +41 -6
- package/package.json +1 -1
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/probe-network.mjs +118 -0
- package/scripts/score-rulings.mjs +107 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +20 -2
- package/src/actions.js +108 -9
- package/src/cli.js +85 -5
- package/src/fingerprint.js +25 -1
- package/src/graph.js +47 -0
- package/src/index.js +41 -0
- package/src/localhelper.js +8 -2
- package/src/mcp.js +16 -7
- package/src/metrics.js +116 -0
- package/src/platform/android.js +23 -0
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +70 -2
- package/src/regions.js +105 -0
- package/src/supervisor.js +51 -7
- package/src/view.js +40 -4
package/src/index.js
CHANGED
|
@@ -303,6 +303,18 @@ export const STALE_FRAME_MS = 2500;
|
|
|
303
303
|
*/
|
|
304
304
|
export const MEMORY_SETTLE_MS = 250;
|
|
305
305
|
|
|
306
|
+
/**
|
|
307
|
+
* How well a relabelled ref must match before it is acted on.
|
|
308
|
+
*
|
|
309
|
+
* Above `matching.MINIMUM_SCORE` (0.45) on purpose: that floor is for a label
|
|
310
|
+
* the caller wrote, and this is a label simframe substituted after refusing
|
|
311
|
+
* their `#n`. A fuzzy name match returns `similarity * 0.72` and a prefix match
|
|
312
|
+
* is scaled by its coverage, so neither can reach 0.8 on the name alone — which
|
|
313
|
+
* makes this "the label matched nearly exactly", not a tuned constant. The
|
|
314
|
+
* recovery that made this necessary scored 0.64.
|
|
315
|
+
*/
|
|
316
|
+
export const RELABEL_MIN_SCORE = 0.8;
|
|
317
|
+
|
|
306
318
|
/** Below this a "change" is a clock digit or a caret, not a new screen. */
|
|
307
319
|
export const MINOR_CHANGE = 0.004;
|
|
308
320
|
export const MAJOR_CHANGE = 0.03;
|
|
@@ -1251,6 +1263,35 @@ async function locateWith(
|
|
|
1251
1263
|
second.staleLabel = err.staleLabel;
|
|
1252
1264
|
throw second;
|
|
1253
1265
|
}
|
|
1266
|
+
// A recovery is held to a higher bar than the lookup it stands in for,
|
|
1267
|
+
// and to one the caller never has to think about.
|
|
1268
|
+
//
|
|
1269
|
+
// `MINIMUM_SCORE` (0.45) is the bar for a label the caller actually
|
|
1270
|
+
// wrote. This label is one *we substituted on their behalf* after
|
|
1271
|
+
// refusing their `#n`, so a weak match here is not "close enough" — it is
|
|
1272
|
+
// us choosing a target nobody named. The reported wrong answer scored
|
|
1273
|
+
// **0.64** and cleared the ordinary floor comfortably.
|
|
1274
|
+
//
|
|
1275
|
+
// 0.8 is structural rather than fitted to that incident: a fuzzy name
|
|
1276
|
+
// match returns `similarity * 0.72` and a prefix match is scaled by its
|
|
1277
|
+
// coverage, so neither reaches 0.8 on the name alone. Only a near-exact
|
|
1278
|
+
// name does. And the region check needs no tuned number at all — if the
|
|
1279
|
+
// map would not offer this target, a recovery may not silently pick it.
|
|
1280
|
+
const region = again.target?.region ?? 'content';
|
|
1281
|
+
const weak = Number.isFinite(again.score) && again.score < RELABEL_MIN_SCORE;
|
|
1282
|
+
if (weak || !regions.offerable(region)) {
|
|
1283
|
+
err.message = `#${selector.ref} cannot be trusted here, and "${err.staleLabel}" was not`
|
|
1284
|
+
+ ' safely re-findable either:'
|
|
1285
|
+
+ (weak ? ` the best match scored ${again.score.toFixed(2)}, below the ${RELABEL_MIN_SCORE} a`
|
|
1286
|
+
+ ' relabel needs (a number you did not ask for may not become a tap on a guess).' : '')
|
|
1287
|
+
+ (!regions.offerable(region) ? ` the best match sits in the ${region}, which sim_ui does not`
|
|
1288
|
+
+ ' offer as something to act on.' : '')
|
|
1289
|
+
+ ' Read the screen again (sim_ui) and name the target.';
|
|
1290
|
+
err.staleRef = true;
|
|
1291
|
+
err.staleLabel = err.staleLabel;
|
|
1292
|
+
err.relabelRefused = { score: again.score ?? null, region };
|
|
1293
|
+
throw err;
|
|
1294
|
+
}
|
|
1254
1295
|
return {
|
|
1255
1296
|
...again,
|
|
1256
1297
|
from: 'ref-relabelled',
|
package/src/localhelper.js
CHANGED
|
@@ -94,7 +94,11 @@ export function lineServer({ ensureBinary, what }) {
|
|
|
94
94
|
try { msg = JSON.parse(line); } catch { continue; }
|
|
95
95
|
if (!settled) {
|
|
96
96
|
settled = true;
|
|
97
|
-
|
|
97
|
+
// The ready line may carry facts about the helper worth keeping —
|
|
98
|
+
// the model's context window, for one, which used to be a constant
|
|
99
|
+
// we repeated in comments. Passed through rather than parsed here,
|
|
100
|
+
// because this file knows about lines and not about models.
|
|
101
|
+
if (msg.ready) resolve({ ok: true, child, waiters, hello: msg });
|
|
98
102
|
else resolve({ ok: false, reason: msg.unavailable ?? `the ${what} did not become ready` });
|
|
99
103
|
continue;
|
|
100
104
|
}
|
|
@@ -145,7 +149,9 @@ export function lineServer({ ensureBinary, what }) {
|
|
|
145
149
|
},
|
|
146
150
|
async status() {
|
|
147
151
|
const live = await open();
|
|
148
|
-
return live?.ok
|
|
152
|
+
return live?.ok
|
|
153
|
+
? { ok: true, hello: live.hello ?? {} }
|
|
154
|
+
: { ok: false, reason: live?.reason ?? 'unavailable' };
|
|
149
155
|
},
|
|
150
156
|
close() {
|
|
151
157
|
try { session?.child?.kill(); } catch { /* already gone */ }
|
package/src/mcp.js
CHANGED
|
@@ -78,13 +78,22 @@ const deviceProp = {
|
|
|
78
78
|
};
|
|
79
79
|
|
|
80
80
|
/**
|
|
81
|
-
* One selector grammar everywhere.
|
|
81
|
+
* One selector grammar everywhere, and the order is the recommendation.
|
|
82
82
|
*
|
|
83
|
-
*
|
|
84
|
-
*
|
|
85
|
-
*
|
|
83
|
+
* It used to lead with `#3` and call it "cheapest and unambiguous". Four peer
|
|
84
|
+
* rounds running reported the opposite: intent resolution worked every time,
|
|
85
|
+
* while refs renumbered underneath them and were only safe inside the round
|
|
86
|
+
* trip that issued them. The README was corrected and these descriptions were
|
|
87
|
+
* not, which is the half a caller actually reads.
|
|
88
|
+
*
|
|
89
|
+
* "Unambiguous" was also the wrong word for it. A ref is exact about which
|
|
90
|
+
* element simframe meant and says nothing about whether that element is still
|
|
91
|
+
* there — the failure mode that needed `staleKind` to tell a moved layout from
|
|
92
|
+
* a different screen, and then a score floor and a region check on top of that
|
|
93
|
+
* before a relabelled ref could be trusted. A label carries its own evidence;
|
|
94
|
+
* a number carries none.
|
|
86
95
|
*/
|
|
87
|
-
const SELECTOR = 'Selector: "#3" (a number from the last screen map —
|
|
96
|
+
const SELECTOR = 'Selector: a label or phrase like "Save" or "the Assets tab" or "back" (resolved by intent — verbs, typos, synonyms, icon-only controls: START HERE), "#3" (a number from the last screen map — exact, but only inside the round trip that numbered it), or "@120,400" for raw point coordinates (last resort: it cannot tell you it missed).';
|
|
88
97
|
|
|
89
98
|
const selectorProp = (what = 'What to act on') => ({
|
|
90
99
|
sel: { type: 'string', description: `${what}. ${SELECTOR}` },
|
|
@@ -94,7 +103,7 @@ const TOOLS = [
|
|
|
94
103
|
{
|
|
95
104
|
name: 'sim_ui',
|
|
96
105
|
description:
|
|
97
|
-
'READ THE SCREEN as text: every element numbered, with region, type, label, state, contents and tap point, plus which screen this is and what simframe knows about it. A tenth the cost of a screenshot and more useful, because it says what is tappable and where.
|
|
106
|
+
'READ THE SCREEN as text: every element numbered, with region, type, label, state, contents and tap point, plus which screen this is and what simframe knows about it. A tenth the cost of a screenshot and more useful, because it says what is tappable and where. Act on what it shows by NAME — whatever it calls "General", you can tap as "General"; the #3 numbers are exact but only until the screen moves. Start here, never with sim_look.',
|
|
98
107
|
inputSchema: {
|
|
99
108
|
type: 'object',
|
|
100
109
|
properties: {
|
|
@@ -123,7 +132,7 @@ const TOOLS = [
|
|
|
123
132
|
steps: {
|
|
124
133
|
type: 'array',
|
|
125
134
|
description:
|
|
126
|
-
'Ordered steps. Every selector below accepts "
|
|
135
|
+
'Ordered steps. Every selector below accepts "Save" | "#3" | "@120,400", in that order of preference. Act: {"tap":"Save"} (add "index" if a label is ambiguous), {"type":{"into":"Name","text":"Fryer 3"}}, {"paste":{"into":"Notes","text":"long text"}}, {"clear":"Notes"} to empty a field and "clear":true on a type/paste to replace rather than append (drop "into" to type into whatever already has focus, which is how you follow a browser next-field chevron — nothing can be read back then, and the step says so), {"scroll":"down"}, {"scrollTo":"Delete account"}, {"swipe":{"from":[x,y],"to":[x,y]}}, {"button":"HOME"}, {"key":"return"} (the keyboard return/enter key, which is how a mobile search field submits — also escape, tab, space, backspace, and the arrows), {"launch":{"value":"com.example.app","relaunch":true,"args":["-uiTest","1"]}}, {"openUrl":"myapp://x"}, {"permission":{"value":"photos","grant":"grant","bundleId":"com.example.app"}}. Check: {"assert":{"value":"Saved","is":"visible"}} (also gone | enabled | disabled | value with "equals"), {"waitFor":{"value":"Saved","timeoutMs":5000}}, {"settle":{"stableMs":600}}, {"pause":300}. Recover without a round trip: add "or" to any step for fallback selectors tried locally — {"tap":"Save","or":["Done","Confirm"]} — and {"seek":"change username","budget":6} explores for something not on this screen: it OPENS containers (a real action — state changes), checks, and returns to where it started, refusing to open anything that commits, abandons or answers. It does not tap the target; it leaves you on the screen where the target resolves so you tap it next. Do not point it into a flow whose progress you cannot afford to lose. A long screen is only knowable a viewport at a time, so {"sweep":"all","fill":{"Last Name":"Asadi","Email":"a@b.c"}} goes to the top, then reads and fills section by section to the bottom — filling each field while it is on screen, which beats finding one and scrolling back. Add "from":"here" to sweep down from where you are. It reports which section each element was in, what it filled, and what it never found at any scroll position. Prefer it to scrollTo on forms and long lists. Brief the supervisor from the plan: top-level "supervise" is standing guidance for the whole batch ("lists here render a count header before rows; REVIEW stays disabled until a provider is chosen") and per-step "expect" adds to it. When it stops a run the result names the steps it did not attempt — re-issue them with a corrected "supervise" note if the judgement was wrong.',
|
|
127
136
|
items: { type: 'object' },
|
|
128
137
|
},
|
|
129
138
|
autoSettle: {
|
package/src/metrics.js
CHANGED
|
@@ -24,6 +24,21 @@ export const REASONS = [
|
|
|
24
24
|
|
|
25
25
|
export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
|
|
26
26
|
|
|
27
|
+
/**
|
|
28
|
+
* What the executor observed after a supervisor ruling — the field that makes a
|
|
29
|
+
* ruling scoreable rather than merely recorded.
|
|
30
|
+
*
|
|
31
|
+
* A ruling on its own says what the supervisor thought. Three items on the
|
|
32
|
+
* deferred list (101's p95 replay, 106's cost of the steps a `stop` skipped,
|
|
33
|
+
* 96's response variable) need what happened *next*, and until now nothing
|
|
34
|
+
* persisted a ruling at all: they went into a `supervisions` array on the
|
|
35
|
+
* result and died with the process. Three rulings had ever existed anywhere.
|
|
36
|
+
*
|
|
37
|
+
* Closed, and mandatory, for the same reason `REASONS` is: a vocabulary that
|
|
38
|
+
* admits "other" collects a pile of "other".
|
|
39
|
+
*/
|
|
40
|
+
export const RULING_OUTCOMES = ['recovered', 'still_failed', 'stopped', 'no_ruling'];
|
|
41
|
+
|
|
27
42
|
/**
|
|
28
43
|
* Which faculty would have removed this escalation.
|
|
29
44
|
*
|
|
@@ -57,6 +72,7 @@ function metricPaths(udid) {
|
|
|
57
72
|
dir,
|
|
58
73
|
escalations: path.join(dir, 'escalations.jsonl'),
|
|
59
74
|
flows: path.join(dir, 'flows.jsonl'),
|
|
75
|
+
supervisions: path.join(dir, 'supervisions.jsonl'),
|
|
60
76
|
baselines: path.join(dir, 'baselines'),
|
|
61
77
|
};
|
|
62
78
|
}
|
|
@@ -109,6 +125,106 @@ export function readJsonl(file, { limit } = {}) {
|
|
|
109
125
|
|
|
110
126
|
export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
|
|
111
127
|
export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
|
|
128
|
+
export const readSupervisions = (udid, opts) => readJsonl(metricPaths(udid).supervisions, opts);
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Write down one supervisor ruling and what came of it.
|
|
132
|
+
*
|
|
133
|
+
* The fields are chosen so the three items waiting on this can be answered
|
|
134
|
+
* **offline, from the log**, rather than by another device run:
|
|
135
|
+
*
|
|
136
|
+
* - `screen`/`edge` — which `(screen_hash, action)` the ruling was about, so
|
|
137
|
+
* rulings can be grouped by edge the way the graph groups timings.
|
|
138
|
+
* - `p95`/`samples` — what the graph knew about that edge **at the moment of
|
|
139
|
+
* the ruling**. Recording it now rather than looking it up later is the
|
|
140
|
+
* difference between 101 being arithmetic and 101 being archaeology: the
|
|
141
|
+
* graph keeps learning, so a p95 read next week is not the p95 the
|
|
142
|
+
* supervisor was implicitly competing with.
|
|
143
|
+
* - `stillMs` — the same input the supervisor got, so a replay sees what it saw.
|
|
144
|
+
* - `expect` — the plan's own note, because a ruling made with a briefing and
|
|
145
|
+
* one made without are not the same measurement (96's critical arm).
|
|
146
|
+
* - `outcome` — what the executor observed afterwards. Mandatory.
|
|
147
|
+
*
|
|
148
|
+
* `from` separates a deterministic rule from a model answer. Both are rulings
|
|
149
|
+
* and both have outcomes, but a comparison that mixed them would credit the
|
|
150
|
+
* model for what a two-line rule decided.
|
|
151
|
+
*/
|
|
152
|
+
export function recordSupervision(udid, {
|
|
153
|
+
session, index, step, edge, screen, decision, from, reason, ms,
|
|
154
|
+
stillMs, p95, samples, expect, failure, outcome,
|
|
155
|
+
}) {
|
|
156
|
+
// Swallowed rather than thrown, unlike `recordEscalation`'s guard, and the
|
|
157
|
+
// difference is deliberate: this is called from inside a flow's failure
|
|
158
|
+
// handler, where a throw would turn a recoverable step failure into a crash.
|
|
159
|
+
// It still lands in `lastWriteError`, which `simframe escalations` prints.
|
|
160
|
+
if (!RULING_OUTCOMES.includes(outcome)) {
|
|
161
|
+
lastWriteError = `supervision outcome must be one of ${RULING_OUTCOMES.join('/')}, got ${JSON.stringify(outcome)}`;
|
|
162
|
+
return false;
|
|
163
|
+
}
|
|
164
|
+
return appendJsonl(metricPaths(udid).supervisions, {
|
|
165
|
+
timestamp: new Date().toISOString(),
|
|
166
|
+
session_id: session ?? SESSION_ID,
|
|
167
|
+
client: CLIENT,
|
|
168
|
+
step_index: index ?? null,
|
|
169
|
+
step: step ?? null,
|
|
170
|
+
edge: edge ?? null,
|
|
171
|
+
screen_fingerprint: screen ?? null,
|
|
172
|
+
decision,
|
|
173
|
+
from: from ?? 'model',
|
|
174
|
+
// Recorded, never presented as the ground for what happened: the supervisor
|
|
175
|
+
// has returned a correct decision with a reason citing a rule that did not
|
|
176
|
+
// apply. Keeping it is how that stays measurable instead of anecdotal.
|
|
177
|
+
reason: reason ? String(reason).slice(0, 200) : null,
|
|
178
|
+
latency_ms: Number.isFinite(ms) ? Math.round(ms) : null,
|
|
179
|
+
still_ms: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
|
|
180
|
+
edge_p95_ms: Number.isFinite(p95) ? Math.round(p95) : null,
|
|
181
|
+
edge_samples: Number.isFinite(samples) ? samples : null,
|
|
182
|
+
expect: expect ? String(expect).slice(0, 200) : null,
|
|
183
|
+
failure: failure ? String(failure).slice(0, 300) : null,
|
|
184
|
+
outcome,
|
|
185
|
+
});
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* What the ruling log currently says, and whether it can yet answer 101.
|
|
190
|
+
*
|
|
191
|
+
* Deliberately counts rather than scores. Item 101 asks which `wait`/`retry`
|
|
192
|
+
* rulings a p95-per-edge lookup would have got right, and that is a separate
|
|
193
|
+
* piece of work; what this answers is the question that comes first and was
|
|
194
|
+
* embarrassing to get wrong once already — **is there a population to measure
|
|
195
|
+
* at all, and do its rows carry the fields the measurement needs.** `p95_known`
|
|
196
|
+
* is that readiness check: a ruling recorded on an edge the graph had never
|
|
197
|
+
* timed cannot take part in the comparison, however many of them there are.
|
|
198
|
+
*/
|
|
199
|
+
export function supervisionBreakdown(records) {
|
|
200
|
+
const by = (key) => {
|
|
201
|
+
const out = {};
|
|
202
|
+
for (const r of records) {
|
|
203
|
+
const k = r[key] ?? 'unknown';
|
|
204
|
+
out[k] = (out[k] ?? 0) + 1;
|
|
205
|
+
}
|
|
206
|
+
return out;
|
|
207
|
+
};
|
|
208
|
+
const matrix = {};
|
|
209
|
+
for (const r of records) {
|
|
210
|
+
const k = `${r.decision ?? 'unknown'} -> ${r.outcome ?? 'unknown'}`;
|
|
211
|
+
matrix[k] = (matrix[k] ?? 0) + 1;
|
|
212
|
+
}
|
|
213
|
+
const timed = records.filter((r) => Number.isFinite(r.edge_p95_ms));
|
|
214
|
+
const latencies = records.map((r) => r.latency_ms).filter((n) => Number.isFinite(n)).sort((a, b) => a - b);
|
|
215
|
+
return {
|
|
216
|
+
total: records.length,
|
|
217
|
+
by_decision: by('decision'),
|
|
218
|
+
by_outcome: by('outcome'),
|
|
219
|
+
by_from: by('from'),
|
|
220
|
+
decision_to_outcome: matrix,
|
|
221
|
+
// Readiness for 101, not a result for it.
|
|
222
|
+
p95_known: timed.length,
|
|
223
|
+
p95_unknown: records.length - timed.length,
|
|
224
|
+
sessions: [...new Set(records.map((r) => r.session_id).filter(Boolean))],
|
|
225
|
+
median_latency_ms: latencies.length ? latencies[latencies.length >> 1] : null,
|
|
226
|
+
};
|
|
227
|
+
}
|
|
112
228
|
|
|
113
229
|
/**
|
|
114
230
|
* Mark an error as an escalation with a reason, at the site that knows why.
|
package/src/platform/android.js
CHANGED
|
@@ -996,6 +996,28 @@ function capabilities() {
|
|
|
996
996
|
};
|
|
997
997
|
}
|
|
998
998
|
|
|
999
|
+
/**
|
|
1000
|
+
* Not implemented here, and it says so in its own vocabulary.
|
|
1001
|
+
*
|
|
1002
|
+
* `revive` exists for a display that has stopped rendering — a CoreSimulator
|
|
1003
|
+
* pathology. An emulator's failure modes are its own (the console port going
|
|
1004
|
+
* away, adb losing the device) and the remedy is not the same sequence, so
|
|
1005
|
+
* borrowing the other platform's answer would be the mistake `doctor` made when
|
|
1006
|
+
* it reported "input driver: idb" for an emulator: a claim about a tool that has
|
|
1007
|
+
* never spoken to an Android device.
|
|
1008
|
+
*
|
|
1009
|
+
* When an emulator wedge is characterised rather than guessed at, this becomes
|
|
1010
|
+
* a real implementation. Until then the honest answer is that this backend does
|
|
1011
|
+
* not have the layer.
|
|
1012
|
+
*/
|
|
1013
|
+
async function restartDevice(serial) {
|
|
1014
|
+
throw new Error(
|
|
1015
|
+
`simframe cannot yet restart an emulator (${serial}) — that remedy is written for a`
|
|
1016
|
+
+ ' simulator display that stopped rendering, and an emulator fails differently.'
|
|
1017
|
+
+ ' Restart it from Android Studio, or `adb -s <serial> emu kill` and relaunch.',
|
|
1018
|
+
);
|
|
1019
|
+
}
|
|
1020
|
+
|
|
999
1021
|
/** @type {import('./index.js').Platform} */
|
|
1000
1022
|
export const platform = {
|
|
1001
1023
|
id: 'android',
|
|
@@ -1011,6 +1033,7 @@ export const platform = {
|
|
|
1011
1033
|
launchApp,
|
|
1012
1034
|
terminateApp,
|
|
1013
1035
|
openUrl,
|
|
1036
|
+
restartDevice,
|
|
1014
1037
|
setPermission,
|
|
1015
1038
|
setPasteboard,
|
|
1016
1039
|
getPasteboard,
|
package/src/platform/index.js
CHANGED
|
@@ -61,7 +61,7 @@ export const PLATFORM_SURFACE = Object.freeze([
|
|
|
61
61
|
'id', 'deviceNoun',
|
|
62
62
|
'listDevices', 'bootedDevices', 'resolveDevice', 'isBootedSync', 'ownsUdid',
|
|
63
63
|
'geometry', 'inputDriver',
|
|
64
|
-
'screenshot', 'launchApp', 'terminateApp', 'openUrl',
|
|
64
|
+
'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
|
|
65
65
|
'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
|
|
66
66
|
'bootedAt',
|
|
67
67
|
]);
|
|
@@ -260,6 +260,16 @@ export const inputDriverFor = (udid) => platformFor(udid).inputDriver(udid);
|
|
|
260
260
|
*/
|
|
261
261
|
export const capabilitiesFor = (udid) => platformFor(udid).capabilities();
|
|
262
262
|
|
|
263
|
+
/**
|
|
264
|
+
* Power-cycle a device, on the backend that owns it.
|
|
265
|
+
*
|
|
266
|
+
* Reached only from `simframe revive`, never from the capture loop: the loop
|
|
267
|
+
* detects a stalled display and reports it, and restarting is the operator's
|
|
268
|
+
* call. A backend that does not have this remedy throws in its own terms rather
|
|
269
|
+
* than borrowing the other's.
|
|
270
|
+
*/
|
|
271
|
+
export const restartDevice = (udid) => platformFor(udid).restartDevice(udid);
|
|
272
|
+
|
|
263
273
|
/** What `simframe doctor` should check: each registered backend's own toolchain. */
|
|
264
274
|
export function toolchainChecks() {
|
|
265
275
|
return backends().flatMap((backend) => backend.toolchain());
|
package/src/platform/ios.js
CHANGED
|
@@ -66,6 +66,19 @@ async function bootedDevices(opts) {
|
|
|
66
66
|
*/
|
|
67
67
|
async function resolveDevice(query, opts) {
|
|
68
68
|
const all = await listDevices(opts);
|
|
69
|
+
return pickDevice(query, all);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Which device a query means, given the whole list.
|
|
74
|
+
*
|
|
75
|
+
* Separated from the listing so the decision can be **tested** rather than
|
|
76
|
+
* reasoned about, because two peer rounds in a row reported being handed the
|
|
77
|
+
* wrong device and both were decided here. The same move as `decisionOf` in the
|
|
78
|
+
* supervisor: the one line where a wrong answer is expensive should be a
|
|
79
|
+
* function somebody can call with adversarial input.
|
|
80
|
+
*/
|
|
81
|
+
export function pickDevice(query, all) {
|
|
69
82
|
const booted = all.filter((d) => d.state === 'Booted');
|
|
70
83
|
if (!query) {
|
|
71
84
|
if (booted.length === 0) throw new Error('no booted simulator (open Simulator.app or run `xcrun simctl boot <udid>`)');
|
|
@@ -98,8 +111,40 @@ async function resolveDevice(query, opts) {
|
|
|
98
111
|
const q = query.toLowerCase();
|
|
99
112
|
const pools = [booted, all];
|
|
100
113
|
for (const pool of pools) {
|
|
101
|
-
|
|
102
|
-
|
|
114
|
+
// A UDID is unique, so an exact UDID match needs no further thought.
|
|
115
|
+
const byUdid = pool.find((d) => d.udid.toLowerCase() === q);
|
|
116
|
+
if (byUdid) return byUdid;
|
|
117
|
+
// A **name is not unique**, and this branch used to treat it as though it
|
|
118
|
+
// were: one `find` over both fields returned whichever device the list
|
|
119
|
+
// happened to put first, and short-circuited past the ambiguity guard
|
|
120
|
+
// below. Item 83 recorded that two booted devices on this machine are both
|
|
121
|
+
// called "iPhone 17 Pro" and answered it by warning in `sim_devices` and
|
|
122
|
+
// printing a UDID prefix in headers — leaving the resolver, which is where
|
|
123
|
+
// the choice is actually made, untouched.
|
|
124
|
+
//
|
|
125
|
+
// What that cost, reported from a three-hour session on a real app: a
|
|
126
|
+
// caller passed the shared name, read a screen that was "38ms old" and an
|
|
127
|
+
// hour wrong, concluded the app had signed itself out, and abandoned a
|
|
128
|
+
// verification run that was fine. The frame was fresh — it was the *other*
|
|
129
|
+
// device's, idling on a login screen. `refresh: true` returned the matching
|
|
130
|
+
// tree because it refreshed the same wrong device. Two independent-looking
|
|
131
|
+
// sources agreeing with each other and both wrong.
|
|
132
|
+
//
|
|
133
|
+
// So a name that names two devices is an ambiguity, exactly like a partial
|
|
134
|
+
// match that hits two, and it refuses for the same reason: the cost of
|
|
135
|
+
// guessing wrong is reading somebody else's screen and believing it.
|
|
136
|
+
const byName = pool.filter((d) => d.name.toLowerCase() === q);
|
|
137
|
+
if (byName.length === 1) return byName[0];
|
|
138
|
+
if (byName.length > 1) {
|
|
139
|
+
throw Object.assign(
|
|
140
|
+
new Error(
|
|
141
|
+
`"${query}" is the name of ${byName.length} devices: `
|
|
142
|
+
+ `${byName.map((d) => d.udid).join(', ')} — a name cannot say which one you mean, `
|
|
143
|
+
+ 'so pass the UDID',
|
|
144
|
+
),
|
|
145
|
+
{ ambiguous: true },
|
|
146
|
+
);
|
|
147
|
+
}
|
|
103
148
|
const partial = pool.filter((d) => d.name.toLowerCase().includes(q));
|
|
104
149
|
if (partial.length === 1) return partial[0];
|
|
105
150
|
if (partial.length > 1) {
|
|
@@ -189,6 +234,28 @@ async function openUrl(udid, url) {
|
|
|
189
234
|
await run('xcrun', ['simctl', 'openurl', udid, url], { timeout: 20_000 });
|
|
190
235
|
}
|
|
191
236
|
|
|
237
|
+
/**
|
|
238
|
+
* Shut a device down and bring it back, waiting for the boot to finish.
|
|
239
|
+
*
|
|
240
|
+
* The remedy for a display that has stopped rendering, which the capture loop
|
|
241
|
+
* can detect and must not perform: it reports `stalled` and stops, because a
|
|
242
|
+
* capture loop that rebooted the device it was watching would be a tool
|
|
243
|
+
* reaching for the mains when a reading looks wrong. This is the operator's
|
|
244
|
+
* decision, reached by `simframe revive`.
|
|
245
|
+
*
|
|
246
|
+
* `bootstatus -b` and not `boot`, for the reason it is used in CI: `boot`
|
|
247
|
+
* returns before the device is usable, and everything downstream then races the
|
|
248
|
+
* boot. Timeboxed generously — a cold boot on a busy machine is slow, and a
|
|
249
|
+
* boot that never finishes should fail here with a reason rather than as a
|
|
250
|
+
* puzzle further down.
|
|
251
|
+
*/
|
|
252
|
+
async function restartDevice(udid) {
|
|
253
|
+
// Tolerated: a device that is already off cannot be shut down, and that is
|
|
254
|
+
// the state this command is most often reached from.
|
|
255
|
+
await run('xcrun', ['simctl', 'shutdown', udid], { timeout: 60_000 }).catch(() => null);
|
|
256
|
+
await run('xcrun', ['simctl', 'bootstatus', udid, '-b'], { timeout: 240_000 });
|
|
257
|
+
}
|
|
258
|
+
|
|
192
259
|
const PERMISSION_SERVICES = [
|
|
193
260
|
'all', 'calendar', 'contacts-limited', 'contacts', 'location', 'location-always',
|
|
194
261
|
'photos-add', 'photos', 'media-library', 'microphone', 'motion', 'reminders', 'siri',
|
|
@@ -347,6 +414,7 @@ export const platform = {
|
|
|
347
414
|
launchApp,
|
|
348
415
|
terminateApp,
|
|
349
416
|
openUrl,
|
|
417
|
+
restartDevice,
|
|
350
418
|
setPermission,
|
|
351
419
|
setPasteboard,
|
|
352
420
|
permissionServices: () => PERMISSION_SERVICES,
|
package/src/regions.js
CHANGED
|
@@ -24,6 +24,27 @@ export const REGIONS = [
|
|
|
24
24
|
'content',
|
|
25
25
|
];
|
|
26
26
|
|
|
27
|
+
/**
|
|
28
|
+
* Regions the screen map will not offer as something to act on.
|
|
29
|
+
*
|
|
30
|
+
* The status bar says the time and the battery level. It is on every screen, it
|
|
31
|
+
* is never what anybody wants to tap, and it costs a row every time — so
|
|
32
|
+
* `sim_ui` hides it.
|
|
33
|
+
*
|
|
34
|
+
* It lives here rather than in `view.js`, where it started, because it is not
|
|
35
|
+
* only a presentation rule. A target the map refuses to *show* must also be a
|
|
36
|
+
* target nothing may resolve onto *behind the caller's back*, and the case that
|
|
37
|
+
* proved it was exactly that: a stale `#1` numbered "Reminders" in Reminders,
|
|
38
|
+
* re-resolved in Contacts onto the status-bar back-to-app breadcrumb
|
|
39
|
+
* "• Reminders", scored 0.64 and handed back a tap point at (47,40) — a place
|
|
40
|
+
* the map would never have put in front of anybody. One rule, one home, both
|
|
41
|
+
* readers.
|
|
42
|
+
*/
|
|
43
|
+
const UNOFFERED_REGIONS = new Set(['status-bar']);
|
|
44
|
+
|
|
45
|
+
/** Would the screen map offer a target in this region? */
|
|
46
|
+
export const offerable = (region) => !UNOFFERED_REGIONS.has(region);
|
|
47
|
+
|
|
27
48
|
/**
|
|
28
49
|
* The status bar stays positional, and deliberately.
|
|
29
50
|
*
|
|
@@ -63,6 +84,46 @@ const BOTTOM_CHROME_LIMIT = 0.82;
|
|
|
63
84
|
* screen sets its own scale.
|
|
64
85
|
*/
|
|
65
86
|
const MIN_BOUNDARY_GAP_PT = 10;
|
|
87
|
+
/**
|
|
88
|
+
* How wide a lone row may be and still be a screen's title rather than its
|
|
89
|
+
* first paragraph.
|
|
90
|
+
*
|
|
91
|
+
* 0.6 of the screen, measured: the Settings root's large title is 133 pt of
|
|
92
|
+
* 402 (0.33), while Reminders' empty-state heading "Welcome to Reminders" is
|
|
93
|
+
* 326 pt (0.81) and is content — it describes the screen instead of naming it.
|
|
94
|
+
* A title is a name and names are short.
|
|
95
|
+
*/
|
|
96
|
+
const LARGE_TITLE_MAX_WIDTH_FRACTION = 0.6;
|
|
97
|
+
/**
|
|
98
|
+
* How far below the status bar a large title can start.
|
|
99
|
+
*
|
|
100
|
+
* iOS draws one at a **system** offset, not an app-chosen one, so this is a
|
|
101
|
+
* bound on a platform constant rather than a tuned threshold. Measured on the
|
|
102
|
+
* bench device: the Settings root's title starts 63-79 pt below the status bar
|
|
103
|
+
* depending on which sensor reports its box, while example.com's `<h1>` — page
|
|
104
|
+
* *content* that merely happens to be the first row, because Safari on iOS puts
|
|
105
|
+
* its chrome at the bottom — starts **122 pt** down.
|
|
106
|
+
*
|
|
107
|
+
* Without this bound the rule promoted that `<h1>` to chrome and "example
|
|
108
|
+
* domain" entered the screen's identity. Pulling page content into identity is
|
|
109
|
+
* the exact failure this module has been bitten by twice (a phantom keyboard,
|
|
110
|
+
* and content that merely fell into a band), so the bound is not optional.
|
|
111
|
+
*/
|
|
112
|
+
const LARGE_TITLE_MAX_INSET_PT = 96;
|
|
113
|
+
/**
|
|
114
|
+
* And the least it can be, before it is just the next row.
|
|
115
|
+
*
|
|
116
|
+
* Absolute, like the maximum, and for the same reason: the inset is drawn by
|
|
117
|
+
* the system, so it is not a function of what the screen contains. The first
|
|
118
|
+
* version tested it against the screen's *median row gap* — and the testbed
|
|
119
|
+
* caught that on its first day, with two screens of the same app. A list of 24
|
|
120
|
+
* rows has a median gap of 0 and the rule fired; a list of 4 rows above a tab
|
|
121
|
+
* bar has a median gap of **414**, because the empty area counts as a gap, and
|
|
122
|
+
* the rule did not. Same title, same inset of 62.9pt, opposite answers — so one
|
|
123
|
+
* screen had a name in its identity and the other did not, and the graph then
|
|
124
|
+
* called them the same screen at 0.50 similarity.
|
|
125
|
+
*/
|
|
126
|
+
const LARGE_TITLE_MIN_INSET_PT = 24;
|
|
66
127
|
const BOUNDARY_GAP_FACTOR = 1.9;
|
|
67
128
|
|
|
68
129
|
/** A tab bar is several things spread across the width, not one thing at the bottom. */
|
|
@@ -177,6 +238,50 @@ export function bands(elements, screen) {
|
|
|
177
238
|
}
|
|
178
239
|
}
|
|
179
240
|
|
|
241
|
+
// --- a large title, which has no gap under it to be found by.
|
|
242
|
+
//
|
|
243
|
+
// The loop above identifies top chrome by the whitespace *beneath* it, and an
|
|
244
|
+
// iOS large title is drawn tight against the content it heads: measured on
|
|
245
|
+
// the Settings root, 79 pt of inset above it and **5.3 pt** below, against a
|
|
246
|
+
// bar of `max(10, typical * 1.9)` = 66.5. No threshold reaches that, so the
|
|
247
|
+
// title fell into `content` and was discarded as content — leaving the screen
|
|
248
|
+
// with **no name at all** in its fingerprint, in either sensor mode.
|
|
249
|
+
//
|
|
250
|
+
// That is not cosmetic. Chrome labels are the only text identity keeps, and
|
|
251
|
+
// `fingerprint.js` names the consequence: two list screens with identical
|
|
252
|
+
// structure differ by their title and nothing else says so. A nameless screen
|
|
253
|
+
// is pure geometry, and on a hosted runner two sparse nameless readings
|
|
254
|
+
// matched exactly — one screen's hash for two screens.
|
|
255
|
+
//
|
|
256
|
+
// So it is found by the inset *above* it instead, which is the half iOS does
|
|
257
|
+
// provide. A large title sits alone, narrow, high, under a generous gap; a
|
|
258
|
+
// compact bar is the mirror image of that (20-33 pt above, 118-134 below) and
|
|
259
|
+
// is already caught by the loop. Deliberately not keyed on the `Heading`
|
|
260
|
+
// role: OCR has no roles, and the reading that actually collided was
|
|
261
|
+
// OCR-only, so a role test would work only in the case that does not fail.
|
|
262
|
+
//
|
|
263
|
+
// Measured across all 17 perception fixtures before being written here: it
|
|
264
|
+
// changes exactly one of them, the Settings root.
|
|
265
|
+
if (!navBarBottom) {
|
|
266
|
+
const first = rows.findIndex((r) => r.top >= statusBarBottom - 1);
|
|
267
|
+
const row = first >= 0 ? rows[first] : null;
|
|
268
|
+
if (
|
|
269
|
+
row
|
|
270
|
+
&& first + 1 < rows.length
|
|
271
|
+
&& row.items.length === 1
|
|
272
|
+
&& (row.items[0].frame?.width ?? 0) <= screen.width * LARGE_TITLE_MAX_WIDTH_FRACTION
|
|
273
|
+
&& row.bottom <= screen.height * TOP_CHROME_LIMIT
|
|
274
|
+
// A real separation from the status bar, but a system-sized one: far
|
|
275
|
+
// enough to be an inset, near enough to still be the app's own title.
|
|
276
|
+
// Both bounds absolute — see LARGE_TITLE_MIN_INSET_PT for what keying the
|
|
277
|
+
// lower one on the screen's own row rhythm cost.
|
|
278
|
+
&& row.top - statusBarBottom >= LARGE_TITLE_MIN_INSET_PT
|
|
279
|
+
&& row.top - statusBarBottom <= LARGE_TITLE_MAX_INSET_PT
|
|
280
|
+
) {
|
|
281
|
+
navBarBottom = row.bottom;
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
|
|
180
285
|
// --- bottom chrome. One row: a tab bar is one row by construction, and the
|
|
181
286
|
// gap above it is what separates it from the list it floats over.
|
|
182
287
|
let tabBarTop = Infinity;
|
package/src/supervisor.js
CHANGED
|
@@ -45,6 +45,30 @@ const helper = lineServer({
|
|
|
45
45
|
|
|
46
46
|
export const DECISIONS = new Set(['wait', 'retry', 'stop']);
|
|
47
47
|
|
|
48
|
+
/**
|
|
49
|
+
* The vocabulary gate, as a function so it can be *tested* rather than grepped.
|
|
50
|
+
*
|
|
51
|
+
* This one line is the safety property: the supervisor cannot invent a step,
|
|
52
|
+
* skip one, substitute a target or continue past an unexpected screen, because
|
|
53
|
+
* those are not words it can say. Anything outside the three is not a decision
|
|
54
|
+
* and becomes `null`, which means "behave as if there is no supervisor".
|
|
55
|
+
*
|
|
56
|
+
* It was previously inline, and the test that guarded it matched the source
|
|
57
|
+
* text — so it broke when the branch grew an else, with nothing actually wrong.
|
|
58
|
+
* A property this important deserves an assertion that runs it.
|
|
59
|
+
*/
|
|
60
|
+
export function decisionOf(answer) {
|
|
61
|
+
// A string, checked rather than coerced. `String(["wait"])` is `"wait"`, so a
|
|
62
|
+
// `String(...)` coercion here let `{decision: ["wait"]}` through the one gate
|
|
63
|
+
// that defines this component's answer space. Found the first time this
|
|
64
|
+
// property was *run* instead of grepped for in the source — the old test
|
|
65
|
+
// matched the source text of the branch and could never have caught it.
|
|
66
|
+
const raw = answer?.decision;
|
|
67
|
+
if (typeof raw !== 'string') return null;
|
|
68
|
+
const decision = raw.toLowerCase();
|
|
69
|
+
return DECISIONS.has(decision) ? decision : null;
|
|
70
|
+
}
|
|
71
|
+
|
|
48
72
|
/** Which backend the caller asked for, per call first and environment second. */
|
|
49
73
|
export function requested(options) {
|
|
50
74
|
const raw = String(options?.supervisor ?? process.env.SIMFRAME_SUPERVISOR ?? '').trim().toLowerCase();
|
|
@@ -62,6 +86,7 @@ export function requested(options) {
|
|
|
62
86
|
*/
|
|
63
87
|
export async function judge({
|
|
64
88
|
goal, step, expected, failure, screen, stillMs, note, options, timeoutMs = 2500,
|
|
89
|
+
detail,
|
|
65
90
|
} = {}) {
|
|
66
91
|
if (!requested(options)) return null;
|
|
67
92
|
if (!step || !failure) return null;
|
|
@@ -74,10 +99,24 @@ export async function judge({
|
|
|
74
99
|
stillMs: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
|
|
75
100
|
note: note ? String(note).slice(0, 200) : null,
|
|
76
101
|
}, timeoutMs);
|
|
77
|
-
const decision =
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
102
|
+
const decision = decisionOf(answer);
|
|
103
|
+
if (decision == null) {
|
|
104
|
+
// Why it did not answer, for the caller's log — through an out-parameter
|
|
105
|
+
// rather than the return value, because returning anything truthy here
|
|
106
|
+
// would change what the executor does. `null` means "behave as if there is
|
|
107
|
+
// no supervisor" and that safety property is the one thing in this file
|
|
108
|
+
// that must not become conditional.
|
|
109
|
+
//
|
|
110
|
+
// Before this, every failure reached the supervision log as "the
|
|
111
|
+
// supervisor did not answer": a timeout, a guardrail refusal and a model
|
|
112
|
+
// that was never installed were one indistinguishable line.
|
|
113
|
+
if (detail && typeof detail === 'object') {
|
|
114
|
+
detail.kind = answer?.kind
|
|
115
|
+
?? (answer == null ? 'no answer' : answer.decision ? 'outside the vocabulary' : 'unparseable');
|
|
116
|
+
if (answer?.error) detail.error = String(answer.error).slice(0, 200);
|
|
117
|
+
}
|
|
118
|
+
return null;
|
|
119
|
+
}
|
|
81
120
|
return { decision, reason: String(answer.reason ?? '').slice(0, 120), ms: answer.ms ?? null };
|
|
82
121
|
}
|
|
83
122
|
|
|
@@ -101,16 +140,21 @@ export async function status(options) {
|
|
|
101
140
|
screen: ['Probe'],
|
|
102
141
|
stillMs: 5000,
|
|
103
142
|
}, 6000);
|
|
104
|
-
|
|
105
|
-
if (!DECISIONS.has(decision)) {
|
|
143
|
+
if (decisionOf(probe) == null) {
|
|
106
144
|
return {
|
|
107
145
|
supervisor: 'none',
|
|
108
146
|
detail: 'the model loaded but did not answer a probe — it is present and not working',
|
|
109
147
|
};
|
|
110
148
|
}
|
|
149
|
+
// The window, read from the model rather than repeated from documentation.
|
|
150
|
+
// Worth printing: the worst case our own clipping allows measures 1,918
|
|
151
|
+
// tokens against it, and that ratio is the reason there is no per-call token
|
|
152
|
+
// budget check — see DEFERRED 99.
|
|
153
|
+
const ctx = live.hello?.contextSize;
|
|
111
154
|
return {
|
|
112
155
|
supervisor: 'apple',
|
|
113
|
-
detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms
|
|
156
|
+
detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms;`
|
|
157
|
+
+ `${ctx ? ` ${ctx}-token window;` : ''} may only answer wait/retry/stop`,
|
|
114
158
|
};
|
|
115
159
|
}
|
|
116
160
|
|