simframe 0.11.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +44 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +29 -1
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +20 -2
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +41 -6
- package/package.json +1 -1
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/probe-network.mjs +118 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +20 -2
- package/src/actions.js +108 -9
- package/src/cli.js +85 -5
- package/src/fingerprint.js +25 -1
- package/src/graph.js +47 -0
- package/src/index.js +41 -0
- package/src/localhelper.js +8 -2
- package/src/mcp.js +16 -7
- package/src/metrics.js +116 -0
- package/src/platform/android.js +23 -0
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +23 -0
- package/src/regions.js +105 -0
- package/src/supervisor.js +51 -7
- package/src/view.js +40 -4
package/src/metrics.js
CHANGED
|
@@ -24,6 +24,21 @@ export const REASONS = [
|
|
|
24
24
|
|
|
25
25
|
export const OUTCOMES = ['resolved_locally', 'escalated_to_model', 'failed'];
|
|
26
26
|
|
|
27
|
+
/**
|
|
28
|
+
* What the executor observed after a supervisor ruling — the field that makes a
|
|
29
|
+
* ruling scoreable rather than merely recorded.
|
|
30
|
+
*
|
|
31
|
+
* A ruling on its own says what the supervisor thought. Three items on the
|
|
32
|
+
* deferred list (101's p95 replay, 106's cost of the steps a `stop` skipped,
|
|
33
|
+
* 96's response variable) need what happened *next*, and until now nothing
|
|
34
|
+
* persisted a ruling at all: they went into a `supervisions` array on the
|
|
35
|
+
* result and died with the process. Three rulings had ever existed anywhere.
|
|
36
|
+
*
|
|
37
|
+
* Closed, and mandatory, for the same reason `REASONS` is: a vocabulary that
|
|
38
|
+
* admits "other" collects a pile of "other".
|
|
39
|
+
*/
|
|
40
|
+
export const RULING_OUTCOMES = ['recovered', 'still_failed', 'stopped', 'no_ruling'];
|
|
41
|
+
|
|
27
42
|
/**
|
|
28
43
|
* Which faculty would have removed this escalation.
|
|
29
44
|
*
|
|
@@ -57,6 +72,7 @@ function metricPaths(udid) {
|
|
|
57
72
|
dir,
|
|
58
73
|
escalations: path.join(dir, 'escalations.jsonl'),
|
|
59
74
|
flows: path.join(dir, 'flows.jsonl'),
|
|
75
|
+
supervisions: path.join(dir, 'supervisions.jsonl'),
|
|
60
76
|
baselines: path.join(dir, 'baselines'),
|
|
61
77
|
};
|
|
62
78
|
}
|
|
@@ -109,6 +125,106 @@ export function readJsonl(file, { limit } = {}) {
|
|
|
109
125
|
|
|
110
126
|
export const readEscalations = (udid, opts) => readJsonl(metricPaths(udid).escalations, opts);
|
|
111
127
|
export const readFlows = (udid, opts) => readJsonl(metricPaths(udid).flows, opts);
|
|
128
|
+
export const readSupervisions = (udid, opts) => readJsonl(metricPaths(udid).supervisions, opts);
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Write down one supervisor ruling and what came of it.
|
|
132
|
+
*
|
|
133
|
+
* The fields are chosen so the three items waiting on this can be answered
|
|
134
|
+
* **offline, from the log**, rather than by another device run:
|
|
135
|
+
*
|
|
136
|
+
* - `screen`/`edge` — which `(screen_hash, action)` the ruling was about, so
|
|
137
|
+
* rulings can be grouped by edge the way the graph groups timings.
|
|
138
|
+
* - `p95`/`samples` — what the graph knew about that edge **at the moment of
|
|
139
|
+
* the ruling**. Recording it now rather than looking it up later is the
|
|
140
|
+
* difference between 101 being arithmetic and 101 being archaeology: the
|
|
141
|
+
* graph keeps learning, so a p95 read next week is not the p95 the
|
|
142
|
+
* supervisor was implicitly competing with.
|
|
143
|
+
* - `stillMs` — the same input the supervisor got, so a replay sees what it saw.
|
|
144
|
+
* - `expect` — the plan's own note, because a ruling made with a briefing and
|
|
145
|
+
* one made without are not the same measurement (96's critical arm).
|
|
146
|
+
* - `outcome` — what the executor observed afterwards. Mandatory.
|
|
147
|
+
*
|
|
148
|
+
* `from` separates a deterministic rule from a model answer. Both are rulings
|
|
149
|
+
* and both have outcomes, but a comparison that mixed them would credit the
|
|
150
|
+
* model for what a two-line rule decided.
|
|
151
|
+
*/
|
|
152
|
+
export function recordSupervision(udid, {
|
|
153
|
+
session, index, step, edge, screen, decision, from, reason, ms,
|
|
154
|
+
stillMs, p95, samples, expect, failure, outcome,
|
|
155
|
+
}) {
|
|
156
|
+
// Swallowed rather than thrown, unlike `recordEscalation`'s guard, and the
|
|
157
|
+
// difference is deliberate: this is called from inside a flow's failure
|
|
158
|
+
// handler, where a throw would turn a recoverable step failure into a crash.
|
|
159
|
+
// It still lands in `lastWriteError`, which `simframe escalations` prints.
|
|
160
|
+
if (!RULING_OUTCOMES.includes(outcome)) {
|
|
161
|
+
lastWriteError = `supervision outcome must be one of ${RULING_OUTCOMES.join('/')}, got ${JSON.stringify(outcome)}`;
|
|
162
|
+
return false;
|
|
163
|
+
}
|
|
164
|
+
return appendJsonl(metricPaths(udid).supervisions, {
|
|
165
|
+
timestamp: new Date().toISOString(),
|
|
166
|
+
session_id: session ?? SESSION_ID,
|
|
167
|
+
client: CLIENT,
|
|
168
|
+
step_index: index ?? null,
|
|
169
|
+
step: step ?? null,
|
|
170
|
+
edge: edge ?? null,
|
|
171
|
+
screen_fingerprint: screen ?? null,
|
|
172
|
+
decision,
|
|
173
|
+
from: from ?? 'model',
|
|
174
|
+
// Recorded, never presented as the ground for what happened: the supervisor
|
|
175
|
+
// has returned a correct decision with a reason citing a rule that did not
|
|
176
|
+
// apply. Keeping it is how that stays measurable instead of anecdotal.
|
|
177
|
+
reason: reason ? String(reason).slice(0, 200) : null,
|
|
178
|
+
latency_ms: Number.isFinite(ms) ? Math.round(ms) : null,
|
|
179
|
+
still_ms: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
|
|
180
|
+
edge_p95_ms: Number.isFinite(p95) ? Math.round(p95) : null,
|
|
181
|
+
edge_samples: Number.isFinite(samples) ? samples : null,
|
|
182
|
+
expect: expect ? String(expect).slice(0, 200) : null,
|
|
183
|
+
failure: failure ? String(failure).slice(0, 300) : null,
|
|
184
|
+
outcome,
|
|
185
|
+
});
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* What the ruling log currently says, and whether it can yet answer 101.
|
|
190
|
+
*
|
|
191
|
+
* Deliberately counts rather than scores. Item 101 asks which `wait`/`retry`
|
|
192
|
+
* rulings a p95-per-edge lookup would have got right, and that is a separate
|
|
193
|
+
* piece of work; what this answers is the question that comes first and was
|
|
194
|
+
* embarrassing to get wrong once already — **is there a population to measure
|
|
195
|
+
* at all, and do its rows carry the fields the measurement needs.** `p95_known`
|
|
196
|
+
* is that readiness check: a ruling recorded on an edge the graph had never
|
|
197
|
+
* timed cannot take part in the comparison, however many of them there are.
|
|
198
|
+
*/
|
|
199
|
+
export function supervisionBreakdown(records) {
|
|
200
|
+
const by = (key) => {
|
|
201
|
+
const out = {};
|
|
202
|
+
for (const r of records) {
|
|
203
|
+
const k = r[key] ?? 'unknown';
|
|
204
|
+
out[k] = (out[k] ?? 0) + 1;
|
|
205
|
+
}
|
|
206
|
+
return out;
|
|
207
|
+
};
|
|
208
|
+
const matrix = {};
|
|
209
|
+
for (const r of records) {
|
|
210
|
+
const k = `${r.decision ?? 'unknown'} -> ${r.outcome ?? 'unknown'}`;
|
|
211
|
+
matrix[k] = (matrix[k] ?? 0) + 1;
|
|
212
|
+
}
|
|
213
|
+
const timed = records.filter((r) => Number.isFinite(r.edge_p95_ms));
|
|
214
|
+
const latencies = records.map((r) => r.latency_ms).filter((n) => Number.isFinite(n)).sort((a, b) => a - b);
|
|
215
|
+
return {
|
|
216
|
+
total: records.length,
|
|
217
|
+
by_decision: by('decision'),
|
|
218
|
+
by_outcome: by('outcome'),
|
|
219
|
+
by_from: by('from'),
|
|
220
|
+
decision_to_outcome: matrix,
|
|
221
|
+
// Readiness for 101, not a result for it.
|
|
222
|
+
p95_known: timed.length,
|
|
223
|
+
p95_unknown: records.length - timed.length,
|
|
224
|
+
sessions: [...new Set(records.map((r) => r.session_id).filter(Boolean))],
|
|
225
|
+
median_latency_ms: latencies.length ? latencies[latencies.length >> 1] : null,
|
|
226
|
+
};
|
|
227
|
+
}
|
|
112
228
|
|
|
113
229
|
/**
|
|
114
230
|
* Mark an error as an escalation with a reason, at the site that knows why.
|
package/src/platform/android.js
CHANGED
|
@@ -996,6 +996,28 @@ function capabilities() {
|
|
|
996
996
|
};
|
|
997
997
|
}
|
|
998
998
|
|
|
999
|
+
/**
|
|
1000
|
+
* Not implemented here, and it says so in its own vocabulary.
|
|
1001
|
+
*
|
|
1002
|
+
* `revive` exists for a display that has stopped rendering — a CoreSimulator
|
|
1003
|
+
* pathology. An emulator's failure modes are its own (the console port going
|
|
1004
|
+
* away, adb losing the device) and the remedy is not the same sequence, so
|
|
1005
|
+
* borrowing the other platform's answer would be the mistake `doctor` made when
|
|
1006
|
+
* it reported "input driver: idb" for an emulator: a claim about a tool that has
|
|
1007
|
+
* never spoken to an Android device.
|
|
1008
|
+
*
|
|
1009
|
+
* When an emulator wedge is characterised rather than guessed at, this becomes
|
|
1010
|
+
* a real implementation. Until then the honest answer is that this backend does
|
|
1011
|
+
* not have the layer.
|
|
1012
|
+
*/
|
|
1013
|
+
async function restartDevice(serial) {
|
|
1014
|
+
throw new Error(
|
|
1015
|
+
`simframe cannot yet restart an emulator (${serial}) — that remedy is written for a`
|
|
1016
|
+
+ ' simulator display that stopped rendering, and an emulator fails differently.'
|
|
1017
|
+
+ ' Restart it from Android Studio, or `adb -s <serial> emu kill` and relaunch.',
|
|
1018
|
+
);
|
|
1019
|
+
}
|
|
1020
|
+
|
|
999
1021
|
/** @type {import('./index.js').Platform} */
|
|
1000
1022
|
export const platform = {
|
|
1001
1023
|
id: 'android',
|
|
@@ -1011,6 +1033,7 @@ export const platform = {
|
|
|
1011
1033
|
launchApp,
|
|
1012
1034
|
terminateApp,
|
|
1013
1035
|
openUrl,
|
|
1036
|
+
restartDevice,
|
|
1014
1037
|
setPermission,
|
|
1015
1038
|
setPasteboard,
|
|
1016
1039
|
getPasteboard,
|
package/src/platform/index.js
CHANGED
|
@@ -61,7 +61,7 @@ export const PLATFORM_SURFACE = Object.freeze([
|
|
|
61
61
|
'id', 'deviceNoun',
|
|
62
62
|
'listDevices', 'bootedDevices', 'resolveDevice', 'isBootedSync', 'ownsUdid',
|
|
63
63
|
'geometry', 'inputDriver',
|
|
64
|
-
'screenshot', 'launchApp', 'terminateApp', 'openUrl',
|
|
64
|
+
'screenshot', 'launchApp', 'terminateApp', 'openUrl', 'restartDevice',
|
|
65
65
|
'setPermission', 'setPasteboard', 'permissionServices', 'capabilities', 'toolchain',
|
|
66
66
|
'bootedAt',
|
|
67
67
|
]);
|
|
@@ -260,6 +260,16 @@ export const inputDriverFor = (udid) => platformFor(udid).inputDriver(udid);
|
|
|
260
260
|
*/
|
|
261
261
|
export const capabilitiesFor = (udid) => platformFor(udid).capabilities();
|
|
262
262
|
|
|
263
|
+
/**
|
|
264
|
+
* Power-cycle a device, on the backend that owns it.
|
|
265
|
+
*
|
|
266
|
+
* Reached only from `simframe revive`, never from the capture loop: the loop
|
|
267
|
+
* detects a stalled display and reports it, and restarting is the operator's
|
|
268
|
+
* call. A backend that does not have this remedy throws in its own terms rather
|
|
269
|
+
* than borrowing the other's.
|
|
270
|
+
*/
|
|
271
|
+
export const restartDevice = (udid) => platformFor(udid).restartDevice(udid);
|
|
272
|
+
|
|
263
273
|
/** What `simframe doctor` should check: each registered backend's own toolchain. */
|
|
264
274
|
export function toolchainChecks() {
|
|
265
275
|
return backends().flatMap((backend) => backend.toolchain());
|
package/src/platform/ios.js
CHANGED
|
@@ -189,6 +189,28 @@ async function openUrl(udid, url) {
|
|
|
189
189
|
await run('xcrun', ['simctl', 'openurl', udid, url], { timeout: 20_000 });
|
|
190
190
|
}
|
|
191
191
|
|
|
192
|
+
/**
|
|
193
|
+
* Shut a device down and bring it back, waiting for the boot to finish.
|
|
194
|
+
*
|
|
195
|
+
* The remedy for a display that has stopped rendering, which the capture loop
|
|
196
|
+
* can detect and must not perform: it reports `stalled` and stops, because a
|
|
197
|
+
* capture loop that rebooted the device it was watching would be a tool
|
|
198
|
+
* reaching for the mains when a reading looks wrong. This is the operator's
|
|
199
|
+
* decision, reached by `simframe revive`.
|
|
200
|
+
*
|
|
201
|
+
* `bootstatus -b` and not `boot`, for the reason it is used in CI: `boot`
|
|
202
|
+
* returns before the device is usable, and everything downstream then races the
|
|
203
|
+
* boot. Timeboxed generously — a cold boot on a busy machine is slow, and a
|
|
204
|
+
* boot that never finishes should fail here with a reason rather than as a
|
|
205
|
+
* puzzle further down.
|
|
206
|
+
*/
|
|
207
|
+
async function restartDevice(udid) {
|
|
208
|
+
// Tolerated: a device that is already off cannot be shut down, and that is
|
|
209
|
+
// the state this command is most often reached from.
|
|
210
|
+
await run('xcrun', ['simctl', 'shutdown', udid], { timeout: 60_000 }).catch(() => null);
|
|
211
|
+
await run('xcrun', ['simctl', 'bootstatus', udid, '-b'], { timeout: 240_000 });
|
|
212
|
+
}
|
|
213
|
+
|
|
192
214
|
const PERMISSION_SERVICES = [
|
|
193
215
|
'all', 'calendar', 'contacts-limited', 'contacts', 'location', 'location-always',
|
|
194
216
|
'photos-add', 'photos', 'media-library', 'microphone', 'motion', 'reminders', 'siri',
|
|
@@ -347,6 +369,7 @@ export const platform = {
|
|
|
347
369
|
launchApp,
|
|
348
370
|
terminateApp,
|
|
349
371
|
openUrl,
|
|
372
|
+
restartDevice,
|
|
350
373
|
setPermission,
|
|
351
374
|
setPasteboard,
|
|
352
375
|
permissionServices: () => PERMISSION_SERVICES,
|
package/src/regions.js
CHANGED
|
@@ -24,6 +24,27 @@ export const REGIONS = [
|
|
|
24
24
|
'content',
|
|
25
25
|
];
|
|
26
26
|
|
|
27
|
+
/**
|
|
28
|
+
* Regions the screen map will not offer as something to act on.
|
|
29
|
+
*
|
|
30
|
+
* The status bar says the time and the battery level. It is on every screen, it
|
|
31
|
+
* is never what anybody wants to tap, and it costs a row every time — so
|
|
32
|
+
* `sim_ui` hides it.
|
|
33
|
+
*
|
|
34
|
+
* It lives here rather than in `view.js`, where it started, because it is not
|
|
35
|
+
* only a presentation rule. A target the map refuses to *show* must also be a
|
|
36
|
+
* target nothing may resolve onto *behind the caller's back*, and the case that
|
|
37
|
+
* proved it was exactly that: a stale `#1` numbered "Reminders" in Reminders,
|
|
38
|
+
* re-resolved in Contacts onto the status-bar back-to-app breadcrumb
|
|
39
|
+
* "• Reminders", scored 0.64 and handed back a tap point at (47,40) — a place
|
|
40
|
+
* the map would never have put in front of anybody. One rule, one home, both
|
|
41
|
+
* readers.
|
|
42
|
+
*/
|
|
43
|
+
const UNOFFERED_REGIONS = new Set(['status-bar']);
|
|
44
|
+
|
|
45
|
+
/** Would the screen map offer a target in this region? */
|
|
46
|
+
export const offerable = (region) => !UNOFFERED_REGIONS.has(region);
|
|
47
|
+
|
|
27
48
|
/**
|
|
28
49
|
* The status bar stays positional, and deliberately.
|
|
29
50
|
*
|
|
@@ -63,6 +84,46 @@ const BOTTOM_CHROME_LIMIT = 0.82;
|
|
|
63
84
|
* screen sets its own scale.
|
|
64
85
|
*/
|
|
65
86
|
const MIN_BOUNDARY_GAP_PT = 10;
|
|
87
|
+
/**
|
|
88
|
+
* How wide a lone row may be and still be a screen's title rather than its
|
|
89
|
+
* first paragraph.
|
|
90
|
+
*
|
|
91
|
+
* 0.6 of the screen, measured: the Settings root's large title is 133 pt of
|
|
92
|
+
* 402 (0.33), while Reminders' empty-state heading "Welcome to Reminders" is
|
|
93
|
+
* 326 pt (0.81) and is content — it describes the screen instead of naming it.
|
|
94
|
+
* A title is a name and names are short.
|
|
95
|
+
*/
|
|
96
|
+
const LARGE_TITLE_MAX_WIDTH_FRACTION = 0.6;
|
|
97
|
+
/**
|
|
98
|
+
* How far below the status bar a large title can start.
|
|
99
|
+
*
|
|
100
|
+
* iOS draws one at a **system** offset, not an app-chosen one, so this is a
|
|
101
|
+
* bound on a platform constant rather than a tuned threshold. Measured on the
|
|
102
|
+
* bench device: the Settings root's title starts 63-79 pt below the status bar
|
|
103
|
+
* depending on which sensor reports its box, while example.com's `<h1>` — page
|
|
104
|
+
* *content* that merely happens to be the first row, because Safari on iOS puts
|
|
105
|
+
* its chrome at the bottom — starts **122 pt** down.
|
|
106
|
+
*
|
|
107
|
+
* Without this bound the rule promoted that `<h1>` to chrome and "example
|
|
108
|
+
* domain" entered the screen's identity. Pulling page content into identity is
|
|
109
|
+
* the exact failure this module has been bitten by twice (a phantom keyboard,
|
|
110
|
+
* and content that merely fell into a band), so the bound is not optional.
|
|
111
|
+
*/
|
|
112
|
+
const LARGE_TITLE_MAX_INSET_PT = 96;
|
|
113
|
+
/**
|
|
114
|
+
* And the least it can be, before it is just the next row.
|
|
115
|
+
*
|
|
116
|
+
* Absolute, like the maximum, and for the same reason: the inset is drawn by
|
|
117
|
+
* the system, so it is not a function of what the screen contains. The first
|
|
118
|
+
* version tested it against the screen's *median row gap* — and the testbed
|
|
119
|
+
* caught that on its first day, with two screens of the same app. A list of 24
|
|
120
|
+
* rows has a median gap of 0 and the rule fired; a list of 4 rows above a tab
|
|
121
|
+
* bar has a median gap of **414**, because the empty area counts as a gap, and
|
|
122
|
+
* the rule did not. Same title, same inset of 62.9pt, opposite answers — so one
|
|
123
|
+
* screen had a name in its identity and the other did not, and the graph then
|
|
124
|
+
* called them the same screen at 0.50 similarity.
|
|
125
|
+
*/
|
|
126
|
+
const LARGE_TITLE_MIN_INSET_PT = 24;
|
|
66
127
|
const BOUNDARY_GAP_FACTOR = 1.9;
|
|
67
128
|
|
|
68
129
|
/** A tab bar is several things spread across the width, not one thing at the bottom. */
|
|
@@ -177,6 +238,50 @@ export function bands(elements, screen) {
|
|
|
177
238
|
}
|
|
178
239
|
}
|
|
179
240
|
|
|
241
|
+
// --- a large title, which has no gap under it to be found by.
|
|
242
|
+
//
|
|
243
|
+
// The loop above identifies top chrome by the whitespace *beneath* it, and an
|
|
244
|
+
// iOS large title is drawn tight against the content it heads: measured on
|
|
245
|
+
// the Settings root, 79 pt of inset above it and **5.3 pt** below, against a
|
|
246
|
+
// bar of `max(10, typical * 1.9)` = 66.5. No threshold reaches that, so the
|
|
247
|
+
// title fell into `content` and was discarded as content — leaving the screen
|
|
248
|
+
// with **no name at all** in its fingerprint, in either sensor mode.
|
|
249
|
+
//
|
|
250
|
+
// That is not cosmetic. Chrome labels are the only text identity keeps, and
|
|
251
|
+
// `fingerprint.js` names the consequence: two list screens with identical
|
|
252
|
+
// structure differ by their title and nothing else says so. A nameless screen
|
|
253
|
+
// is pure geometry, and on a hosted runner two sparse nameless readings
|
|
254
|
+
// matched exactly — one screen's hash for two screens.
|
|
255
|
+
//
|
|
256
|
+
// So it is found by the inset *above* it instead, which is the half iOS does
|
|
257
|
+
// provide. A large title sits alone, narrow, high, under a generous gap; a
|
|
258
|
+
// compact bar is the mirror image of that (20-33 pt above, 118-134 below) and
|
|
259
|
+
// is already caught by the loop. Deliberately not keyed on the `Heading`
|
|
260
|
+
// role: OCR has no roles, and the reading that actually collided was
|
|
261
|
+
// OCR-only, so a role test would work only in the case that does not fail.
|
|
262
|
+
//
|
|
263
|
+
// Measured across all 17 perception fixtures before being written here: it
|
|
264
|
+
// changes exactly one of them, the Settings root.
|
|
265
|
+
if (!navBarBottom) {
|
|
266
|
+
const first = rows.findIndex((r) => r.top >= statusBarBottom - 1);
|
|
267
|
+
const row = first >= 0 ? rows[first] : null;
|
|
268
|
+
if (
|
|
269
|
+
row
|
|
270
|
+
&& first + 1 < rows.length
|
|
271
|
+
&& row.items.length === 1
|
|
272
|
+
&& (row.items[0].frame?.width ?? 0) <= screen.width * LARGE_TITLE_MAX_WIDTH_FRACTION
|
|
273
|
+
&& row.bottom <= screen.height * TOP_CHROME_LIMIT
|
|
274
|
+
// A real separation from the status bar, but a system-sized one: far
|
|
275
|
+
// enough to be an inset, near enough to still be the app's own title.
|
|
276
|
+
// Both bounds absolute — see LARGE_TITLE_MIN_INSET_PT for what keying the
|
|
277
|
+
// lower one on the screen's own row rhythm cost.
|
|
278
|
+
&& row.top - statusBarBottom >= LARGE_TITLE_MIN_INSET_PT
|
|
279
|
+
&& row.top - statusBarBottom <= LARGE_TITLE_MAX_INSET_PT
|
|
280
|
+
) {
|
|
281
|
+
navBarBottom = row.bottom;
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
|
|
180
285
|
// --- bottom chrome. One row: a tab bar is one row by construction, and the
|
|
181
286
|
// gap above it is what separates it from the list it floats over.
|
|
182
287
|
let tabBarTop = Infinity;
|
package/src/supervisor.js
CHANGED
|
@@ -45,6 +45,30 @@ const helper = lineServer({
|
|
|
45
45
|
|
|
46
46
|
export const DECISIONS = new Set(['wait', 'retry', 'stop']);
|
|
47
47
|
|
|
48
|
+
/**
|
|
49
|
+
* The vocabulary gate, as a function so it can be *tested* rather than grepped.
|
|
50
|
+
*
|
|
51
|
+
* This one line is the safety property: the supervisor cannot invent a step,
|
|
52
|
+
* skip one, substitute a target or continue past an unexpected screen, because
|
|
53
|
+
* those are not words it can say. Anything outside the three is not a decision
|
|
54
|
+
* and becomes `null`, which means "behave as if there is no supervisor".
|
|
55
|
+
*
|
|
56
|
+
* It was previously inline, and the test that guarded it matched the source
|
|
57
|
+
* text — so it broke when the branch grew an else, with nothing actually wrong.
|
|
58
|
+
* A property this important deserves an assertion that runs it.
|
|
59
|
+
*/
|
|
60
|
+
export function decisionOf(answer) {
|
|
61
|
+
// A string, checked rather than coerced. `String(["wait"])` is `"wait"`, so a
|
|
62
|
+
// `String(...)` coercion here let `{decision: ["wait"]}` through the one gate
|
|
63
|
+
// that defines this component's answer space. Found the first time this
|
|
64
|
+
// property was *run* instead of grepped for in the source — the old test
|
|
65
|
+
// matched the source text of the branch and could never have caught it.
|
|
66
|
+
const raw = answer?.decision;
|
|
67
|
+
if (typeof raw !== 'string') return null;
|
|
68
|
+
const decision = raw.toLowerCase();
|
|
69
|
+
return DECISIONS.has(decision) ? decision : null;
|
|
70
|
+
}
|
|
71
|
+
|
|
48
72
|
/** Which backend the caller asked for, per call first and environment second. */
|
|
49
73
|
export function requested(options) {
|
|
50
74
|
const raw = String(options?.supervisor ?? process.env.SIMFRAME_SUPERVISOR ?? '').trim().toLowerCase();
|
|
@@ -62,6 +86,7 @@ export function requested(options) {
|
|
|
62
86
|
*/
|
|
63
87
|
export async function judge({
|
|
64
88
|
goal, step, expected, failure, screen, stillMs, note, options, timeoutMs = 2500,
|
|
89
|
+
detail,
|
|
65
90
|
} = {}) {
|
|
66
91
|
if (!requested(options)) return null;
|
|
67
92
|
if (!step || !failure) return null;
|
|
@@ -74,10 +99,24 @@ export async function judge({
|
|
|
74
99
|
stillMs: Number.isFinite(stillMs) ? Math.round(stillMs) : null,
|
|
75
100
|
note: note ? String(note).slice(0, 200) : null,
|
|
76
101
|
}, timeoutMs);
|
|
77
|
-
const decision =
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
102
|
+
const decision = decisionOf(answer);
|
|
103
|
+
if (decision == null) {
|
|
104
|
+
// Why it did not answer, for the caller's log — through an out-parameter
|
|
105
|
+
// rather than the return value, because returning anything truthy here
|
|
106
|
+
// would change what the executor does. `null` means "behave as if there is
|
|
107
|
+
// no supervisor" and that safety property is the one thing in this file
|
|
108
|
+
// that must not become conditional.
|
|
109
|
+
//
|
|
110
|
+
// Before this, every failure reached the supervision log as "the
|
|
111
|
+
// supervisor did not answer": a timeout, a guardrail refusal and a model
|
|
112
|
+
// that was never installed were one indistinguishable line.
|
|
113
|
+
if (detail && typeof detail === 'object') {
|
|
114
|
+
detail.kind = answer?.kind
|
|
115
|
+
?? (answer == null ? 'no answer' : answer.decision ? 'outside the vocabulary' : 'unparseable');
|
|
116
|
+
if (answer?.error) detail.error = String(answer.error).slice(0, 200);
|
|
117
|
+
}
|
|
118
|
+
return null;
|
|
119
|
+
}
|
|
81
120
|
return { decision, reason: String(answer.reason ?? '').slice(0, 120), ms: answer.ms ?? null };
|
|
82
121
|
}
|
|
83
122
|
|
|
@@ -101,16 +140,21 @@ export async function status(options) {
|
|
|
101
140
|
screen: ['Probe'],
|
|
102
141
|
stillMs: 5000,
|
|
103
142
|
}, 6000);
|
|
104
|
-
|
|
105
|
-
if (!DECISIONS.has(decision)) {
|
|
143
|
+
if (decisionOf(probe) == null) {
|
|
106
144
|
return {
|
|
107
145
|
supervisor: 'none',
|
|
108
146
|
detail: 'the model loaded but did not answer a probe — it is present and not working',
|
|
109
147
|
};
|
|
110
148
|
}
|
|
149
|
+
// The window, read from the model rather than repeated from documentation.
|
|
150
|
+
// Worth printing: the worst case our own clipping allows measures 1,918
|
|
151
|
+
// tokens against it, and that ratio is the reason there is no per-call token
|
|
152
|
+
// budget check — see DEFERRED 99.
|
|
153
|
+
const ctx = live.hello?.contextSize;
|
|
111
154
|
return {
|
|
112
155
|
supervisor: 'apple',
|
|
113
|
-
detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms
|
|
156
|
+
detail: `Apple Foundation Models, on-device; answered a probe in ${probe.ms ?? '?'}ms;`
|
|
157
|
+
+ `${ctx ? ` ${ctx}-token window;` : ''} may only answer wait/retry/stop`,
|
|
114
158
|
};
|
|
115
159
|
}
|
|
116
160
|
|
package/src/view.js
CHANGED
|
@@ -22,10 +22,10 @@ import * as matching from './matching.js';
|
|
|
22
22
|
const REGION_ORDER = ['nav-bar', 'content', 'tab-bar', 'keyboard', 'status-bar'];
|
|
23
23
|
|
|
24
24
|
/**
|
|
25
|
-
*
|
|
26
|
-
*
|
|
25
|
+
* Which regions the map will not offer now lives in `regions.js`, because it is
|
|
26
|
+
* not only a presentation rule — see `regions.offerable`. A target this hides
|
|
27
|
+
* must also be one nothing resolves onto behind the caller's back.
|
|
27
28
|
*/
|
|
28
|
-
const HIDDEN_REGIONS = new Set(['status-bar']);
|
|
29
29
|
|
|
30
30
|
/** A keyboard is 30-odd keys nobody refers to by name. One line says it. */
|
|
31
31
|
const COLLAPSE_REGIONS = new Set(['keyboard']);
|
|
@@ -249,7 +249,7 @@ export function rowsFor(entry, { screen, filter, interactive, all = false, limit
|
|
|
249
249
|
if (!isNum(t.x) || !isNum(t.y)) return false;
|
|
250
250
|
// Off-screen elements are real in the tree and untappable in fact.
|
|
251
251
|
if (regions.offViewport(t, screen)) return false;
|
|
252
|
-
if (!all &&
|
|
252
|
+
if (!all && !regions.offerable(t.region)) return false;
|
|
253
253
|
if (!all && isNoise(t)) return false;
|
|
254
254
|
return true;
|
|
255
255
|
});
|
|
@@ -329,6 +329,28 @@ export function recalledNote(identity, now = Date.now()) {
|
|
|
329
329
|
return `elements recalled from ${ago} ago — pass refresh for what is there now`;
|
|
330
330
|
}
|
|
331
331
|
|
|
332
|
+
/**
|
|
333
|
+
* How old the frame behind this reading is, and whether that is a problem.
|
|
334
|
+
*
|
|
335
|
+
* Two thresholds, because "slightly behind" and "possibly a different screen"
|
|
336
|
+
* are different messages. Under `FRAME_FRESH_MS` nothing is said: a map that
|
|
337
|
+
* announced "42ms old" on every call would train a reader to skip the line that
|
|
338
|
+
* matters. Over `FRAME_STALE_MS` it shouts, because at that age the app may have
|
|
339
|
+
* moved on entirely and the whole element list is then a description of the past.
|
|
340
|
+
*/
|
|
341
|
+
export const FRAME_FRESH_MS = 1500;
|
|
342
|
+
export const FRAME_STALE_MS = 4000;
|
|
343
|
+
|
|
344
|
+
export function frameAgeNote(identity, now = Date.now()) {
|
|
345
|
+
const at = identity?.state?.capturedAt;
|
|
346
|
+
if (!Number.isFinite(at)) return null;
|
|
347
|
+
const age = Math.max(0, now - at);
|
|
348
|
+
if (age < FRAME_FRESH_MS) return null;
|
|
349
|
+
if (age < FRAME_STALE_MS) return `frame ${(age / 1000).toFixed(1)}s old`;
|
|
350
|
+
return `WARNING this frame is ${(age / 1000).toFixed(1)}s old — the screen may have moved on `
|
|
351
|
+
+ 'since, so treat the elements below as a description of the past and pass refresh';
|
|
352
|
+
}
|
|
353
|
+
|
|
332
354
|
/**
|
|
333
355
|
* What a control *contains*, from the sensor that actually knows.
|
|
334
356
|
*
|
|
@@ -702,6 +724,20 @@ export function render({ device, identity, rows, truncated, collapsed, screen, n
|
|
|
702
724
|
identity?.settled === false ? 'STILL MOVING' : null,
|
|
703
725
|
// Still and finished are not the same thing.
|
|
704
726
|
identity?.loading === true ? 'STILL LOADING' : null,
|
|
727
|
+
// How old the *frame* this map was read from is.
|
|
728
|
+
//
|
|
729
|
+
// `sim_look` and `sim_state` have printed this since they existed, and this
|
|
730
|
+
// map never has — so the one tool an agent is told to start with was the one
|
|
731
|
+
// that could not say how old its evidence was. Reported from the field: a
|
|
732
|
+
// complete 20-element map of a screen the app was not on, and *"a wrong
|
|
733
|
+
// answer is worse than an error here, because nothing downstream knows to
|
|
734
|
+
// doubt it"*. `ensureDaemon` will hand back a frame up to 30s old, so this
|
|
735
|
+
// was reachable without anything being broken.
|
|
736
|
+
//
|
|
737
|
+
// `recalledNote` below is a different claim — that the *element map* came
|
|
738
|
+
// from memory — and having one was what made the absence of the other easy
|
|
739
|
+
// to miss.
|
|
740
|
+
frameAgeNote(identity),
|
|
705
741
|
recalledNote(identity),
|
|
706
742
|
].filter(Boolean).join(' · ');
|
|
707
743
|
|