simframe 0.11.0 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +44 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +29 -1
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +20 -2
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +41 -6
- package/package.json +1 -1
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/probe-network.mjs +118 -0
- package/scripts/score-rulings.mjs +107 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +20 -2
- package/src/actions.js +108 -9
- package/src/cli.js +85 -5
- package/src/fingerprint.js +25 -1
- package/src/graph.js +47 -0
- package/src/index.js +41 -0
- package/src/localhelper.js +8 -2
- package/src/mcp.js +16 -7
- package/src/metrics.js +116 -0
- package/src/platform/android.js +23 -0
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +70 -2
- package/src/regions.js +105 -0
- package/src/supervisor.js +51 -7
- package/src/view.js +40 -4
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Score a ruling population against the baselines that could embarrass it.
|
|
3
|
+
//
|
|
4
|
+
// Written **before** the population it first scored was finished, deliberately.
|
|
5
|
+
// The previous attempt computed a `stillMs` threshold after seeing the answers,
|
|
6
|
+
// on 14 samples of which 12 shared one label, and produced "100%" — a number
|
|
7
|
+
// that was fitted, not measured. Fixing that afterwards is not possible: you
|
|
8
|
+
// cannot un-see the data. So the split rule, the candidate thresholds and the
|
|
9
|
+
// baselines are all fixed here in advance.
|
|
10
|
+
//
|
|
11
|
+
// node scripts/score-rulings.mjs --device=<udid>
|
|
12
|
+
//
|
|
13
|
+
// What it will not do: pick the best threshold and report its score. It fits on
|
|
14
|
+
// the first half and scores on the second, and prints both numbers so a gap
|
|
15
|
+
// between them is visible as overfitting rather than hidden as success.
|
|
16
|
+
import * as metrics from '../src/metrics.js';
|
|
17
|
+
import { resolveDevice } from '../src/platform/index.js';
|
|
18
|
+
|
|
19
|
+
const arg = (n, d) => {
|
|
20
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
21
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
/** The fixtures, and the word each one's situation actually calls for. */
|
|
25
|
+
const CORRECT = {
|
|
26
|
+
'the list is still loading; its rows arrive shortly after launch': 'wait',
|
|
27
|
+
'the detail screen is still fetching; its text arrives shortly': 'wait',
|
|
28
|
+
'the list arrives in waves and this row is in the last one': 'wait',
|
|
29
|
+
'Review is blocked until Species is filled in, and it is empty': 'stop',
|
|
30
|
+
'the first submit always fails and the second works, so waiting cannot help': 'stop',
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
/** `retry` and `wait` differ only in how long they settle, so both satisfy a wait. */
|
|
34
|
+
const satisfies = (decision, want) =>
|
|
35
|
+
(want === 'wait' ? decision === 'wait' || decision === 'retry' : decision === want);
|
|
36
|
+
|
|
37
|
+
const dev = await resolveDevice(arg('device'));
|
|
38
|
+
const all = metrics.readSupervisions(dev.udid).filter((r) => CORRECT[r.expect]);
|
|
39
|
+
// `--last=N` scores one batch rather than the whole log. Needed the first time
|
|
40
|
+
// this ran: the log still held rulings from a deliberately skewed population,
|
|
41
|
+
// so the balance read 19/35 and the baseline 65% when the batch actually under
|
|
42
|
+
// test was 16/16. Mixing a known-bad population into the denominator is the
|
|
43
|
+
// same error as before wearing a different hat.
|
|
44
|
+
const lastN = Number(arg('last', 0));
|
|
45
|
+
const rows = lastN > 0 ? all.slice(-lastN) : all;
|
|
46
|
+
if (rows.length < 8) {
|
|
47
|
+
console.error(`only ${rows.length} labelled ruling(s) — run scripts/collect-rulings.mjs first`);
|
|
48
|
+
process.exit(2);
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const want = (r) => CORRECT[r.expect];
|
|
52
|
+
const counts = rows.reduce((a, r) => ({ ...a, [want(r)]: (a[want(r)] ?? 0) + 1 }), {});
|
|
53
|
+
const commonest = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
|
|
54
|
+
const baseline = commonest[1] / rows.length;
|
|
55
|
+
|
|
56
|
+
console.log(`${rows.length} labelled rulings on ${dev.name}`);
|
|
57
|
+
console.log(`balance: ${JSON.stringify(counts)}`);
|
|
58
|
+
console.log(`majority-class baseline: always "${commonest[0]}" scores ${(100 * baseline).toFixed(0)}%`);
|
|
59
|
+
if (baseline > 0.65) {
|
|
60
|
+
console.log('\nSKEWED — any accuracy below is mostly a fact about the fixture set.');
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const acc = (rs, predict) => rs.filter((r) => satisfies(predict(r), want(r))).length / (rs.length || 1);
|
|
64
|
+
const pct = (x) => `${(100 * x).toFixed(0)}%`;
|
|
65
|
+
|
|
66
|
+
console.log(`\n${'the model'.padEnd(30)} ${pct(acc(rows, (r) => r.decision))}`);
|
|
67
|
+
console.log(`${`always "${commonest[0]}"`.padEnd(30)} ${pct(baseline)}`);
|
|
68
|
+
|
|
69
|
+
// --- the stillMs rule, fitted on one half and scored on the other.
|
|
70
|
+
//
|
|
71
|
+
// Split by *recording order* rather than at random or by fixture, because the
|
|
72
|
+
// alternative is choosing a split, and choosing is the thing that went wrong.
|
|
73
|
+
const half = Math.floor(rows.length / 2);
|
|
74
|
+
const fit = rows.slice(0, half);
|
|
75
|
+
const held = rows.slice(half);
|
|
76
|
+
const CANDIDATES = [1000, 1500, 2000, 2500, 3000, 4000, 5000, 6000, 7000];
|
|
77
|
+
const rule = (t) => (r) => ((r.still_ms ?? 0) > t ? 'stop' : 'wait');
|
|
78
|
+
|
|
79
|
+
let best = null;
|
|
80
|
+
for (const t of CANDIDATES) {
|
|
81
|
+
const a = acc(fit, rule(t));
|
|
82
|
+
if (!best || a > best.a) best = { t, a };
|
|
83
|
+
}
|
|
84
|
+
console.log(`\nstillMs threshold, fitted on the first ${fit.length} and scored on the last ${held.length}:`);
|
|
85
|
+
console.log(` chosen on the fit half: > ${best.t}ms (${pct(best.a)} there)`);
|
|
86
|
+
console.log(` ${'on the held-out half:'.padEnd(24)} ${pct(acc(held, rule(best.t)))}`);
|
|
87
|
+
console.log(` ${'the model, same half:'.padEnd(24)} ${pct(acc(held, (r) => r.decision))}`);
|
|
88
|
+
console.log('\nA rule that scores far better on the fit half than the held-out half was');
|
|
89
|
+
console.log('fitted to noise. That gap is the number this script exists to print.');
|
|
90
|
+
|
|
91
|
+
// --- the direction of the errors, which does not depend on the balance at all.
|
|
92
|
+
const wrong = rows.filter((r) => !satisfies(r.decision, want(r)));
|
|
93
|
+
const dirs = wrong.reduce((a, r) => {
|
|
94
|
+
const k = `said "${r.decision}" where "${want(r)}" was right`;
|
|
95
|
+
return { ...a, [k]: (a[k] ?? 0) + 1 };
|
|
96
|
+
}, {});
|
|
97
|
+
console.log(`\n${wrong.length} error(s) of ${rows.length}:`);
|
|
98
|
+
for (const [k, n] of Object.entries(dirs).sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${k}`);
|
|
99
|
+
if (Object.keys(dirs).length === 1 && wrong.length > 2) {
|
|
100
|
+
console.log('\nAll errors in one direction. A one-directional bias is what an abstain');
|
|
101
|
+
console.log('token addresses (DEFERRED 100), whatever the headline accuracy says.');
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const lat = rows.map((r) => r.latency_ms).filter(Number.isFinite).sort((a, b) => a - b);
|
|
105
|
+
if (lat.length) console.log(`\nmedian judgement latency: ${lat[lat.length >> 1]}ms`);
|
|
106
|
+
const timed = rows.filter((r) => Number.isFinite(r.edge_p95_ms)).length;
|
|
107
|
+
console.log(`edges the graph had timed: ${timed}/${rows.length}`);
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// How long can this device be driven before capture wedges?
|
|
3
|
+
//
|
|
4
|
+
// The wedge is the thing standing between us and a ruling population: the
|
|
5
|
+
// display stops rendering after a few minutes of hard driving, `simctl
|
|
6
|
+
// screenshot` fails too, both of the daemon's recoveries fail, and only a
|
|
7
|
+
// device restart cures it. Three times in one afternoon.
|
|
8
|
+
//
|
|
9
|
+
// A leak was found in our own code on 2026-09-12 — damage-callback registration
|
|
10
|
+
// was not idempotent, and the recovery loop called it on every attempt, so one
|
|
11
|
+
// log's 670 port re-resolves meant up to 670 live callbacks on one port, each
|
|
12
|
+
// invoked per redraw. That is a real leak with a plausible path to saturating
|
|
13
|
+
// the display service, and it is **not proof** that it is the cause. This is
|
|
14
|
+
// how we find out: drive until it dies, and report how long that took.
|
|
15
|
+
//
|
|
16
|
+
// node scripts/soak-capture.mjs --device=<udid> --minutes=25
|
|
17
|
+
//
|
|
18
|
+
// A number to compare against, not a pass/fail. Before the fix, the device
|
|
19
|
+
// wedged roughly every 10-20 minutes of this kind of work.
|
|
20
|
+
import * as actions from '../src/actions.js';
|
|
21
|
+
import * as api from '../src/index.js';
|
|
22
|
+
import * as store from '../src/store.js';
|
|
23
|
+
|
|
24
|
+
const arg = (n, d) => {
|
|
25
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
26
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
27
|
+
};
|
|
28
|
+
const device = arg('device');
|
|
29
|
+
const minutes = Number(arg('minutes', 20));
|
|
30
|
+
const BUNDLE = 'com.example.simframetestbed';
|
|
31
|
+
|
|
32
|
+
const { device: dev } = await api.ensureDaemon(device);
|
|
33
|
+
console.log(`device: ${dev.name} (${dev.runtime}) budget: ${minutes} minutes`);
|
|
34
|
+
|
|
35
|
+
const started = Date.now();
|
|
36
|
+
const deadline = started + minutes * 60_000;
|
|
37
|
+
let laps = 0;
|
|
38
|
+
let reads = 0;
|
|
39
|
+
const elapsed = () => ((Date.now() - started) / 60_000).toFixed(1);
|
|
40
|
+
|
|
41
|
+
// A lap is deliberately the kind of work that provokes it: launching, moving
|
|
42
|
+
// between screens, and a cold read on each — not idling with a poll.
|
|
43
|
+
const LAP = [
|
|
44
|
+
[{ launch: { value: BUNDLE, relaunch: true } }, { pause: 2500 }],
|
|
45
|
+
[{ tap: 'Forms, tab, 2 of 3' }, { pause: 800 }],
|
|
46
|
+
[{ tap: 'Long form' }, { pause: 900 }],
|
|
47
|
+
[{ scroll: 'down' }, { pause: 500 }],
|
|
48
|
+
[{ scroll: 'down' }, { pause: 500 }],
|
|
49
|
+
[{ tap: 'Plants, tab, 1 of 3' }, { pause: 900 }],
|
|
50
|
+
[{ tap: 'Diagnostics, tab, 3 of 3' }, { pause: 900 }],
|
|
51
|
+
];
|
|
52
|
+
|
|
53
|
+
while (Date.now() < deadline) {
|
|
54
|
+
for (const steps of LAP) {
|
|
55
|
+
try {
|
|
56
|
+
await actions.runScript(device, { steps, verify: false, options: { supervisor: 'none' } });
|
|
57
|
+
} catch { /* a failed step is not the subject; a dead display is */ }
|
|
58
|
+
try {
|
|
59
|
+
await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
60
|
+
reads += 1;
|
|
61
|
+
} catch (err) {
|
|
62
|
+
console.log(`\nWEDGED after ${elapsed()} minutes, ${laps} laps, ${reads} cold reads`);
|
|
63
|
+
console.log(` ${String(err.message).split('\n')[0]}`);
|
|
64
|
+
const health = store.captureHealth(dev.udid);
|
|
65
|
+
if (health) console.log(` captureHealth: ${JSON.stringify(health)}`);
|
|
66
|
+
process.exit(1);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
laps += 1;
|
|
70
|
+
if (laps % 5 === 0) process.stdout.write(` ${elapsed()}m ${laps} laps ${reads} reads still alive\n`);
|
|
71
|
+
}
|
|
72
|
+
console.log(`\nSURVIVED ${minutes} minutes: ${laps} laps, ${reads} cold reads, no wedge`);
|
package/skills/simframe/SKILL.md
CHANGED
|
@@ -230,7 +230,10 @@ Measured against the same screen as an image: **~460 tokens of text versus
|
|
|
230
230
|
image path degrades to base64-as-text. The text also says what is *tappable*
|
|
231
231
|
and where, which an image does not.
|
|
232
232
|
|
|
233
|
-
The numbers are selectors
|
|
233
|
+
The numbers are selectors, and so are the labels beside them. Act by **name** —
|
|
234
|
+
whatever `ui` calls `Weekly digest`, you can tap as `"Weekly digest"`. A `#3`
|
|
235
|
+
is exact but only until the screen moves; see Selectors below for why that
|
|
236
|
+
order is a correction rather than a preference.
|
|
234
237
|
|
|
235
238
|
**Reach for an image only when the text genuinely cannot answer the question:**
|
|
236
239
|
visual layout, colour, spacing, an animation, or something neither the
|
|
@@ -246,7 +249,7 @@ a baseline captured *before* it, so steps cannot race the UI.
|
|
|
246
249
|
cat > /tmp/flow.json <<'JSON'
|
|
247
250
|
[{"tap": "Inbox tab"},
|
|
248
251
|
{"assert": {"value": "Weekly digest", "is": "visible"}},
|
|
249
|
-
{"tap": "
|
|
252
|
+
{"tap": "Weekly digest"},
|
|
250
253
|
{"type": {"into": "Reply", "text": "on it"}},
|
|
251
254
|
{"scrollTo": "Send"},
|
|
252
255
|
{"tap": "Send"},
|
|
@@ -312,6 +315,9 @@ the next call should change.
|
|
|
312
315
|
| `#N cannot be trusted here — … this is a different screen` | the screen really did change. Read it again |
|
|
313
316
|
| `autoSettle was off, so the map below was read without waiting` | the map may describe the screen *before* the last action landed |
|
|
314
317
|
| `"iPhone 17 Pro" names more than one booted device` | a name cannot identify which device answered. Pass `device` with a UDID |
|
|
318
|
+
| `frame 2.5s old` | the map was read from a frame that old. Usually fine after a deliberate pause; worth noticing if you expected to have just acted |
|
|
319
|
+
| `WARNING this frame is Ns old — … a description of the past` | the screen may have moved on entirely. **Pass `refresh`** before believing any element below it. Reported from the field against 0.11.0: a complete 20-element map of a screen the app was not on, with nothing saying so |
|
|
320
|
+
| `capture is wedged and both recoveries are spent` | the simulator has stopped rendering and neither of simframe's recoveries helped. Run `simframe revive --device=<udid>`; nothing else will work until you do |
|
|
315
321
|
|
|
316
322
|
## Every step is verified, and the verdict means something
|
|
317
323
|
|
|
@@ -388,8 +394,19 @@ as one call adds one exchange to the context instead of twelve.
|
|
|
388
394
|
```bash
|
|
389
395
|
simframe doctor # capture engine, input driver, a11y, OCR — each honestly
|
|
390
396
|
simframe doctor --strict # any degraded layer is a non-zero exit
|
|
397
|
+
simframe input reset # rebuild the HID session, without restarting anything
|
|
398
|
+
simframe revive # power-cycle a device that has stopped rendering
|
|
391
399
|
```
|
|
392
400
|
|
|
401
|
+
**If capture has wedged**, `doctor` says so and nothing below it will work. A
|
|
402
|
+
simulator driven hard for several minutes can stop rendering — `simctl
|
|
403
|
+
screenshot` fails too, so it is the device and not simframe's view of it — and
|
|
404
|
+
the daemon tries two recoveries, reports the state, and stops there. `simframe
|
|
405
|
+
revive` does the restart in the order that matters and confirms frames are
|
|
406
|
+
flowing again afterwards. It is a command rather than a behaviour on purpose:
|
|
407
|
+
a capture loop that rebooted the device it was watching would be a tool
|
|
408
|
+
reaching for the mains.
|
|
409
|
+
|
|
393
410
|
simframe falls back when it must — the simctl capture loop instead of the
|
|
394
411
|
daemon, idb instead of the in-process input and accessibility paths — but it
|
|
395
412
|
never falls back quietly. If
|
|
@@ -405,6 +422,7 @@ simframe recall # what happened in the last ~60s, as text
|
|
|
405
422
|
simframe strip # recent frames tiled into one image, for an animation
|
|
406
423
|
simframe find "the save button" # resolve an intent without acting on it
|
|
407
424
|
simframe wait --mode=settle # block until the screen stops reacting
|
|
425
|
+
simframe supervisions # what the local supervisor decided, and what came of it
|
|
408
426
|
```
|
|
409
427
|
|
|
410
428
|
`recall` matters more than it looks: if you look up and the screen is already
|
package/src/actions.js
CHANGED
|
@@ -209,6 +209,62 @@ export async function runScript(
|
|
|
209
209
|
/* instrumentation must not be able to fail a flow it is only watching */
|
|
210
210
|
}
|
|
211
211
|
};
|
|
212
|
+
/**
|
|
213
|
+
* What the caller is shown for each outcome, keyed by what the log stores.
|
|
214
|
+
*
|
|
215
|
+
* Two vocabularies on purpose. The log wants tokens that will still parse in
|
|
216
|
+
* six months; the result and the CLI line want English, and a test already
|
|
217
|
+
* pins those words. Mapping them here is what stops the two drifting.
|
|
218
|
+
*/
|
|
219
|
+
const SHOWN = {
|
|
220
|
+
recovered: 'recovered',
|
|
221
|
+
still_failed: 'still failed',
|
|
222
|
+
stopped: 'stopped the run',
|
|
223
|
+
no_ruling: 'no ruling',
|
|
224
|
+
};
|
|
225
|
+
/**
|
|
226
|
+
* Record one ruling, in memory for the caller and on disk for the analysis.
|
|
227
|
+
*
|
|
228
|
+
* One helper rather than four call sites because the two had already drifted
|
|
229
|
+
* apart once — `from` was on the `stop` push and on none of the others, so a
|
|
230
|
+
* rule-sourced wait was indistinguishable from a model-sourced one in the
|
|
231
|
+
* only place that reported them. A ruling that reaches the caller and not the
|
|
232
|
+
* log is the state item 101a exists to end: three rulings had ever existed
|
|
233
|
+
* anywhere, and 101, 106 and 96 are all waiting on a population.
|
|
234
|
+
*/
|
|
235
|
+
const noteRuling = (index, ruling, outcome, { step, expect, failure, why } = {}) => {
|
|
236
|
+
const shown = SHOWN[outcome] ?? outcome;
|
|
237
|
+
const unanswered = why?.kind
|
|
238
|
+
? `the supervisor did not answer: ${why.kind}`
|
|
239
|
+
: 'the supervisor did not answer';
|
|
240
|
+
supervisions.push({
|
|
241
|
+
index,
|
|
242
|
+
decision: ruling?.decision ?? 'unavailable',
|
|
243
|
+
reason: ruling?.reason ?? unanswered,
|
|
244
|
+
from: ruling?.from,
|
|
245
|
+
outcome: shown,
|
|
246
|
+
});
|
|
247
|
+
try {
|
|
248
|
+
metrics.recordSupervision(udid, {
|
|
249
|
+
index,
|
|
250
|
+
step: step ? graph.actionSignature(step) : ruling?.context?.action ?? null,
|
|
251
|
+
edge: ruling?.context?.edge ?? null,
|
|
252
|
+
screen: ruling?.context?.screen ?? null,
|
|
253
|
+
decision: ruling?.decision ?? 'unavailable',
|
|
254
|
+
from: ruling?.from ?? (ruling ? 'model' : 'none'),
|
|
255
|
+
reason: ruling?.reason ?? unanswered,
|
|
256
|
+
ms: ruling?.ms,
|
|
257
|
+
stillMs: ruling?.context?.stillMs,
|
|
258
|
+
p95: ruling?.context?.p95,
|
|
259
|
+
samples: ruling?.context?.samples,
|
|
260
|
+
expect,
|
|
261
|
+
failure,
|
|
262
|
+
outcome,
|
|
263
|
+
});
|
|
264
|
+
} catch {
|
|
265
|
+
/* instrumentation must not be able to fail a flow it is only watching */
|
|
266
|
+
}
|
|
267
|
+
};
|
|
212
268
|
|
|
213
269
|
const needsInput = steps.some((s) => ACTION_STEPS.has(normalizeStep(s).action));
|
|
214
270
|
if (needsInput) {
|
|
@@ -309,8 +365,15 @@ export async function runScript(
|
|
|
309
365
|
// Ask the supervisor before anything is abandoned. It sits behind the
|
|
310
366
|
// hands and in front of the reasoner: first responder, not
|
|
311
367
|
// decision-maker, and its whole vocabulary is wait/retry/stop.
|
|
368
|
+
// Why a consultation produced nothing, for the log. Every failure used
|
|
369
|
+
// to arrive as "the supervisor did not answer" — a timeout, a guardrail
|
|
370
|
+
// refusal and a model that was never installed reading identically.
|
|
371
|
+
const why = {};
|
|
312
372
|
const ruling = await superviseFailure(deviceQuery, {
|
|
313
|
-
goal: supervise ?? flowName, step, expected: step.expect, err, options,
|
|
373
|
+
goal: supervise ?? flowName, step, expected: step.expect, err, options, udid, detail: why,
|
|
374
|
+
});
|
|
375
|
+
const ruled = (outcome) => noteRuling(i, ruling, outcome, {
|
|
376
|
+
step, expect: step.expect, failure: err.message, why,
|
|
314
377
|
});
|
|
315
378
|
if (ruling?.decision === 'wait' || ruling?.decision === 'retry') {
|
|
316
379
|
// Both wait and retry settle first, differing only in how long.
|
|
@@ -332,10 +395,10 @@ export async function runScript(
|
|
|
332
395
|
try {
|
|
333
396
|
detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
|
|
334
397
|
detail += ` [the local supervisor said ${ruling.decision}; it worked on the second attempt]`;
|
|
335
|
-
|
|
398
|
+
ruled('recovered');
|
|
336
399
|
continue;
|
|
337
400
|
} catch (again) {
|
|
338
|
-
|
|
401
|
+
ruled('still_failed');
|
|
339
402
|
err = again;
|
|
340
403
|
}
|
|
341
404
|
} else if (!ruling && supervisor.requested(options)) {
|
|
@@ -345,10 +408,12 @@ export async function runScript(
|
|
|
345
408
|
// model healthy. The reporter's own words — *"had I run flow 2 alone
|
|
346
409
|
// I would have reported the supervisor makes no difference without
|
|
347
410
|
// realising it had never run"*.
|
|
348
|
-
|
|
349
|
-
err.message += ' — the local supervisor was consulted and did not answer
|
|
411
|
+
ruled('no_ruling');
|
|
412
|
+
err.message += ' — the local supervisor was consulted and did not answer'
|
|
413
|
+
+ (why.kind ? ` (${why.kind})` : '')
|
|
414
|
+
+ ', so this failure was not judged.';
|
|
350
415
|
} else if (ruling?.decision === 'stop') {
|
|
351
|
-
|
|
416
|
+
ruled('stopped');
|
|
352
417
|
const remaining = steps.slice(i);
|
|
353
418
|
err.message += ` — the local supervisor stopped the run here.`
|
|
354
419
|
+ ` ${remaining.length} step(s) were not attempted.`
|
|
@@ -1123,14 +1188,34 @@ const SUPERVISOR_RETRY_MS = 900;
|
|
|
1123
1188
|
/** A scroll moves at once or not at all; it does not need a transition's budget. */
|
|
1124
1189
|
const SCROLL_SETTLE_MS = 800;
|
|
1125
1190
|
|
|
1126
|
-
async function superviseFailure(deviceQuery, { goal, step, expected, err, options }) {
|
|
1191
|
+
async function superviseFailure(deviceQuery, { goal, step, expected, err, options, udid, detail }) {
|
|
1127
1192
|
if (!supervisor.requested(options)) return null;
|
|
1193
|
+
// The action half of the edge is free — the step is in hand — so a rule-sourced
|
|
1194
|
+
// ruling still records which action it was about even though it never reads
|
|
1195
|
+
// the screen. Deliberately no `screenMap` call on this path: the rule exists
|
|
1196
|
+
// to answer without one, and paying for a read to make the log prettier would
|
|
1197
|
+
// slow the fast path to improve the bookkeeping.
|
|
1198
|
+
const action = graph.actionSignature(step);
|
|
1128
1199
|
const settled = deterministicRuling(err);
|
|
1129
|
-
if (settled)
|
|
1200
|
+
if (settled) {
|
|
1201
|
+
return {
|
|
1202
|
+
decision: settled.decision, reason: settled.why, from: 'rule',
|
|
1203
|
+
context: { action, edge: action, screen: null, stillMs: null, p95: null, samples: null },
|
|
1204
|
+
};
|
|
1205
|
+
}
|
|
1130
1206
|
try {
|
|
1131
1207
|
const map = await view.screenMap(deviceQuery, { options, refresh: false });
|
|
1132
1208
|
const stillMs = map.identity?.state?.motion?.stillForMs;
|
|
1133
|
-
|
|
1209
|
+
const screen = map.identity?.hash ?? null;
|
|
1210
|
+
// What the graph knew about this edge at the moment of the ruling. Read
|
|
1211
|
+
// here and not at analysis time on purpose: the graph keeps learning, so a
|
|
1212
|
+
// p95 looked up next week is not the number this ruling was competing
|
|
1213
|
+
// with, and item 101 asks exactly "would a p95 lookup have got this right".
|
|
1214
|
+
let timing = null;
|
|
1215
|
+
try {
|
|
1216
|
+
timing = udid && map.identity ? graph.timingFor(udid, map.identity, step) : null;
|
|
1217
|
+
} catch { /* an unknown edge is a fact about the graph, not a failure here */ }
|
|
1218
|
+
const ruling = await supervisor.judge({
|
|
1134
1219
|
goal,
|
|
1135
1220
|
step: `${step.action} ${JSON.stringify(String(step.value ?? step.target ?? step.into ?? step.seek ?? '').slice(0, 60))}`,
|
|
1136
1221
|
expected,
|
|
@@ -1139,7 +1224,21 @@ async function superviseFailure(deviceQuery, { goal, step, expected, err, option
|
|
|
1139
1224
|
stillMs,
|
|
1140
1225
|
note: stillFillingIn(map.identity?.entry),
|
|
1141
1226
|
options,
|
|
1227
|
+
detail,
|
|
1142
1228
|
});
|
|
1229
|
+
if (!ruling) return null;
|
|
1230
|
+
return {
|
|
1231
|
+
...ruling,
|
|
1232
|
+
from: ruling.from ?? 'model',
|
|
1233
|
+
context: {
|
|
1234
|
+
action,
|
|
1235
|
+
edge: screen ? `${String(screen).slice(0, 12)}:${action}` : action,
|
|
1236
|
+
screen,
|
|
1237
|
+
stillMs,
|
|
1238
|
+
p95: timing?.p95 ?? null,
|
|
1239
|
+
samples: timing?.samples ?? null,
|
|
1240
|
+
},
|
|
1241
|
+
};
|
|
1143
1242
|
} catch {
|
|
1144
1243
|
return null;
|
|
1145
1244
|
}
|
package/src/cli.js
CHANGED
|
@@ -3,7 +3,7 @@ import fs from 'node:fs';
|
|
|
3
3
|
import os from 'node:os';
|
|
4
4
|
import path from 'node:path';
|
|
5
5
|
import { runDaemon, DEFAULTS } from './daemon.js';
|
|
6
|
-
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, screenshot, toolchainChecks } from './platform/index.js';
|
|
6
|
+
import { bootedDevices, capabilitiesFor, listDevices, PLATFORMS, resolveDevice, restartDevice, screenshot, toolchainChecks } from './platform/index.js';
|
|
7
7
|
import * as actions from './actions.js';
|
|
8
8
|
import * as analyze from './analyze.js';
|
|
9
9
|
import * as api from './index.js';
|
|
@@ -47,6 +47,8 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
47
47
|
simframe baseline list recorded runs per flow, and what is committed
|
|
48
48
|
simframe hpi [device] Human Parity Index, per flow and overall
|
|
49
49
|
simframe escalations [device] why simframe handed decisions back, by reason
|
|
50
|
+
simframe supervisions [device] local supervisor rulings, and what came of each
|
|
51
|
+
simframe revive [device] power-cycle a wedged device: stop, shutdown, boot, start, reset input
|
|
50
52
|
(--session=<id> narrows to one agent; the
|
|
51
53
|
ids are listed in the output. SIMFRAME_SESSION
|
|
52
54
|
names one, but only at process start — an
|
|
@@ -56,10 +58,12 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
56
58
|
(--strict, or SIMFRAME_STRICT=1, makes any
|
|
57
59
|
degraded layer a non-zero exit)
|
|
58
60
|
|
|
59
|
-
Selectors — anywhere a control is named
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
61
|
+
Selectors — anywhere a control is named, best first
|
|
62
|
+
"Save" a label or a phrase, resolved by intent (verbs, typos, synonyms,
|
|
63
|
+
icon-only controls by their common name). Start here.
|
|
64
|
+
#3 the number \`simframe ui\` gave it. Exact, but only inside the
|
|
65
|
+
round trip that numbered it — the screen moves and it does not.
|
|
66
|
+
@120,400 raw point coordinates. Last resort: it cannot tell you it missed.
|
|
63
67
|
|
|
64
68
|
Measuring against a human — the Human Parity Index
|
|
65
69
|
|
|
@@ -378,6 +382,51 @@ async function main() {
|
|
|
378
382
|
return;
|
|
379
383
|
}
|
|
380
384
|
|
|
385
|
+
// The power cycle, when the narrow remedies are spent.
|
|
386
|
+
//
|
|
387
|
+
// Deliberately a command and not a behaviour. The capture loop tries two
|
|
388
|
+
// things — re-resolve the display port, then rebind the device — and then
|
|
389
|
+
// reports `stalled` and stops, because a capture loop that rebooted the
|
|
390
|
+
// device it was watching would be a tool reaching for the mains when a
|
|
391
|
+
// reading looks wrong. Restarting is the operator's call.
|
|
392
|
+
//
|
|
393
|
+
// But it was the operator's call *and* their four commands, remembered from
|
|
394
|
+
// a handoff note: stop the daemon, shut the device down, boot it and wait,
|
|
395
|
+
// start capture, rebuild the HID session. Done by hand three times in one
|
|
396
|
+
// afternoon, in that order, because any other order leaves a daemon holding
|
|
397
|
+
// a dead device. So the tool knows the order now; the decision is still
|
|
398
|
+
// yours.
|
|
399
|
+
case 'revive': {
|
|
400
|
+
const dev = await resolveDevice(device);
|
|
401
|
+
const say = (line) => { if (!flags.json) console.log(line); };
|
|
402
|
+
const steps = [];
|
|
403
|
+
const did = async (what, fn) => {
|
|
404
|
+
try { await fn(); steps.push({ step: what, ok: true }); say(` ok ${what}`); } catch (err) {
|
|
405
|
+
steps.push({ step: what, ok: false, error: err.message });
|
|
406
|
+
say(` .. ${what} — ${err.message.split('\n')[0]}`);
|
|
407
|
+
}
|
|
408
|
+
};
|
|
409
|
+
say(`reviving ${dev.name}`);
|
|
410
|
+
// Forced: the point of this command is that the device is wedged, so
|
|
411
|
+
// something is certainly still holding it.
|
|
412
|
+
await did('stopped the daemon', async () => { api.stopDaemon(dev.udid, { force: true }); });
|
|
413
|
+
// Through the boundary, which is the whole point of the boundary: the
|
|
414
|
+
// first version of this shelled out to `xcrun` from here and the test
|
|
415
|
+
// that forbids it failed immediately, correctly.
|
|
416
|
+
await did('restarted the device, and waited for the boot to finish',
|
|
417
|
+
() => restartDevice(dev.udid));
|
|
418
|
+
await did('started capture', () => api.ensureDaemon(dev.udid));
|
|
419
|
+
await did('rebuilt the HID session', () => input.resetSession(dev.udid));
|
|
420
|
+
const health = await api.getState(dev.udid).then((s) => s?.state ?? null).catch(() => null);
|
|
421
|
+
const alive = Boolean(health?.hash);
|
|
422
|
+
emit(flags, { ok: alive, device: dev.udid, steps }, alive
|
|
423
|
+
? `\n${dev.name} is producing frames again`
|
|
424
|
+
: `\n${dev.name} is still not producing frames. This is past what simframe can do —`
|
|
425
|
+
+ ' check Simulator.app is not showing an error, and see docs/DEFERRED.md item 95.');
|
|
426
|
+
if (!alive) process.exitCode = 1;
|
|
427
|
+
return;
|
|
428
|
+
}
|
|
429
|
+
|
|
381
430
|
case 'status': {
|
|
382
431
|
const udids = device
|
|
383
432
|
? [(await resolveDevice(device)).udid]
|
|
@@ -1027,6 +1076,37 @@ async function main() {
|
|
|
1027
1076
|
return;
|
|
1028
1077
|
}
|
|
1029
1078
|
|
|
1079
|
+
case 'supervisions': {
|
|
1080
|
+
const dev = await resolveDevice(flags.device);
|
|
1081
|
+
const records = metrics.readSupervisions(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
|
|
1082
|
+
const b = metrics.supervisionBreakdown(records);
|
|
1083
|
+
if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify({ ...b, records }, null, 2)}\n`);
|
|
1084
|
+
emit(flags, { ...b, records: flags.verbose ? records : undefined }, [
|
|
1085
|
+
`${b.total} supervisor ruling${b.total === 1 ? '' : 's'} on ${dev.name}`,
|
|
1086
|
+
b.total ? '' : 'Nothing has been judged on this device yet. The supervisor is off unless'
|
|
1087
|
+
+ ' SIMFRAME_SUPERVISOR=apple, and a ruling is only recorded when a step actually fails.',
|
|
1088
|
+
...Object.entries(b.decision_to_outcome)
|
|
1089
|
+
.sort((a, c) => c[1] - a[1])
|
|
1090
|
+
.map(([k, n]) => ` ${k.padEnd(28)} ${String(n).padStart(4)}`),
|
|
1091
|
+
b.total ? '' : null,
|
|
1092
|
+
b.total ? `sourced: ${Object.entries(b.by_from).map(([k, n]) => `${k} ${n}`).join(', ')}` : null,
|
|
1093
|
+
b.median_latency_ms != null ? `median latency: ${b.median_latency_ms}ms` : null,
|
|
1094
|
+
// Said out loud, because the first version of item 101 claimed its
|
|
1095
|
+
// measurement ran "on logs we already have" when nothing persisted a
|
|
1096
|
+
// ruling at all. This line is what stops that claim being made twice.
|
|
1097
|
+
b.total
|
|
1098
|
+
? `edges the graph had timed: ${b.p95_known}/${b.total}`
|
|
1099
|
+
+ (b.p95_unknown
|
|
1100
|
+
? ` — ${b.p95_unknown} ruling(s) are on edges with no p95, so they cannot take part in 101's comparison`
|
|
1101
|
+
: '')
|
|
1102
|
+
: null,
|
|
1103
|
+
b.sessions.length > 1
|
|
1104
|
+
? `WARNING ${b.sessions.length} sessions are pooled here; two agents on one device write one file`
|
|
1105
|
+
: null,
|
|
1106
|
+
].filter((l) => l !== null).join('\n'));
|
|
1107
|
+
break;
|
|
1108
|
+
}
|
|
1109
|
+
|
|
1030
1110
|
case 'escalations': {
|
|
1031
1111
|
const dev = await resolveDevice(flags.device);
|
|
1032
1112
|
const records = metrics.readEscalations(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
|
package/src/fingerprint.js
CHANGED
|
@@ -48,8 +48,32 @@ import * as regions from './regions.js';
|
|
|
48
48
|
* extends upward while the rows above keep being key-shaped, which page
|
|
49
49
|
* content is not. Any screen fingerprinted with a keyboard up hashes
|
|
50
50
|
* differently, so the stored graph and maps must go again.
|
|
51
|
+
*
|
|
52
|
+
* 7 — a screen whose own name is an iOS **large title** had no name at all in
|
|
53
|
+
* its identity. Chrome labels are the only text these tokens keep, and a
|
|
54
|
+
* large title is drawn tight against the content it heads — 79 pt of inset
|
|
55
|
+
* above it and 5.3 pt below, against a boundary bar of 66.5 — so the top
|
|
56
|
+
* chrome detector, which looks for the gap *beneath* a bar, never found it
|
|
57
|
+
* and the title was discarded as content. Measured on the Settings root in
|
|
58
|
+
* both sensor modes: **0 named tokens**, and the same for Contacts and
|
|
59
|
+
* Reminders. On a hosted runner two sparse nameless readings then matched
|
|
60
|
+
* exactly and one hash stood for two different screens. `regions.bands`
|
|
61
|
+
* now finds a large title by the inset above it; every affected screen
|
|
62
|
+
* hashes differently, so stored graphs and maps go again.
|
|
63
|
+
*
|
|
64
|
+
* 8 — the same rule, keyed on the wrong thing. Its "is this a real inset"
|
|
65
|
+
* test compared the gap above the title against the screen's **median row
|
|
66
|
+
* gap**, so whether a system-drawn title counted as chrome depended on how
|
|
67
|
+
* many rows happened to sit below it. Caught on the React Native testbed's
|
|
68
|
+
* first day, with two screens of one app: a list of 24 rows has a median
|
|
69
|
+
* gap of 0 and the rule fired, a list of 4 above a tab bar has a median gap
|
|
70
|
+
* of **414** and it did not. Same title, same 62.9pt inset, opposite
|
|
71
|
+
* answers — so one screen carried a name and the other did not, and the
|
|
72
|
+
* graph merged them. Both bounds are absolute now. Screens that were
|
|
73
|
+
* missed at 7 hash differently at 8, and unlike a stale hash that matches
|
|
74
|
+
* nothing, these matched the *wrong* thing.
|
|
51
75
|
*/
|
|
52
|
-
export const TOKEN_RULES_VERSION =
|
|
76
|
+
export const TOKEN_RULES_VERSION = 8;
|
|
53
77
|
|
|
54
78
|
/** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
|
|
55
79
|
export const GRID = 24;
|
package/src/graph.js
CHANGED
|
@@ -258,6 +258,28 @@ export function nearestScreen(udid, screen, { threshold = SIMILARITY_THRESHOLD }
|
|
|
258
258
|
for (const node of nodes) {
|
|
259
259
|
for (const f of fingerprintsOf(node)) {
|
|
260
260
|
if (!f.tokens?.length) continue;
|
|
261
|
+
// A screen that says it is called something else is not this screen.
|
|
262
|
+
//
|
|
263
|
+
// Similarity weighs every token equally, and a chrome label is not an
|
|
264
|
+
// equal token — it is the only text in a fingerprint and the only thing
|
|
265
|
+
// that distinguishes two screens of the same shape. `fingerprint.js` has
|
|
266
|
+
// said so in a comment since it was written: "two list screens with
|
|
267
|
+
// identical structure differ by their title, and nothing else says so".
|
|
268
|
+
// Nothing enforced it.
|
|
269
|
+
//
|
|
270
|
+
// Measured on the RN testbed, which is what found this: a list of plants
|
|
271
|
+
// and a list of forms, each a large title over full-width rows above a
|
|
272
|
+
// tab bar, scored **0.50** against a 0.36 threshold and became one node.
|
|
273
|
+
// Both were correctly named by then; the name was simply outvoted, being
|
|
274
|
+
// one token of a union of six. The graph then offered one screen's
|
|
275
|
+
// controls on the other, which is the failure `verify` exists to stop.
|
|
276
|
+
//
|
|
277
|
+
// Both sides must actually carry names for this to apply. A reading whose
|
|
278
|
+
// tree did not answer has no names through no fault of the screen's, and
|
|
279
|
+
// the two sensors are already known to disagree about 0.33-0.47 of a
|
|
280
|
+
// token set — so "one has names, the other does not" is a fact about the
|
|
281
|
+
// sensors and must not be read as a fact about identity.
|
|
282
|
+
if (disagreeOnName(f.tokens, key.tokens)) continue;
|
|
261
283
|
const s = fingerprint.similarity(f.tokens, key.tokens);
|
|
262
284
|
if (s > bestSimilarity) {
|
|
263
285
|
bestSimilarity = s;
|
|
@@ -268,6 +290,31 @@ export function nearestScreen(udid, screen, { threshold = SIMILARITY_THRESHOLD }
|
|
|
268
290
|
return best && bestSimilarity >= threshold ? { node: best, similarity: bestSimilarity } : null;
|
|
269
291
|
}
|
|
270
292
|
|
|
293
|
+
/** The chrome labels in a token set — the only text a fingerprint keeps. */
|
|
294
|
+
function namesIn(tokens) {
|
|
295
|
+
const out = new Set();
|
|
296
|
+
for (const t of tokens ?? []) {
|
|
297
|
+
const open = t.indexOf('"');
|
|
298
|
+
if (open < 0) continue;
|
|
299
|
+
const close = t.lastIndexOf('"');
|
|
300
|
+
if (close > open) out.add(t.slice(open + 1, close));
|
|
301
|
+
}
|
|
302
|
+
return out;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/**
|
|
306
|
+
* Do these two readings name themselves, and name themselves differently?
|
|
307
|
+
*
|
|
308
|
+
* Only a positive disagreement counts. Silence on either side is not evidence.
|
|
309
|
+
*/
|
|
310
|
+
function disagreeOnName(a, b) {
|
|
311
|
+
const left = namesIn(a);
|
|
312
|
+
const right = namesIn(b);
|
|
313
|
+
if (!left.size || !right.size) return false;
|
|
314
|
+
for (const name of left) if (right.has(name)) return false;
|
|
315
|
+
return true;
|
|
316
|
+
}
|
|
317
|
+
|
|
271
318
|
/**
|
|
272
319
|
* Teach a node that it also looks like this.
|
|
273
320
|
*
|