simframe 0.13.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -0
- package/native/simframed/Sources/SimframeCore/Motion.swift +33 -2
- package/package.json +1 -1
- package/scripts/ci-device-guard.mjs +82 -0
- package/scripts/ci-integration-local.sh +11 -8
- package/scripts/ci-memory.mjs +25 -1
- package/scripts/eval-fingerprint.mjs +125 -2
- package/src/actions.js +117 -4
- package/src/cli.js +23 -2
- package/src/graph.js +15 -2
- package/src/matching.js +17 -1
- package/src/mcp.js +141 -12
- package/src/navigate.js +4 -1
- package/src/platform/android.js +34 -0
- package/src/platform/index.js +7 -0
- package/src/platform/ios.js +98 -0
- package/src/platform/plist.js +156 -0
- package/src/storage.js +201 -0
package/README.md
CHANGED
|
@@ -406,6 +406,7 @@ steer the model is a tool surface the model uses wrong.
|
|
|
406
406
|
| `sim_wait` | Waits for the screen to change *and then* settle. |
|
|
407
407
|
| `sim_look` | **The only tool that returns an image**, capped at 1024 px. For layout, colour, spacing — questions text cannot answer. |
|
|
408
408
|
| `sim_recall` · `sim_strip` | Look backwards: a text timeline of what happened, or recent frames tiled into one image. |
|
|
409
|
+
| `sim_storage` | **What the app believes**, as opposed to what it drew: its `UserDefaults` and, for React Native, its `AsyncStorage`. Reads the data container off disk, so it answers on a device that is **not running**. |
|
|
409
410
|
| `sim_capture` · `sim_devices` | Manage capture loops; list simulators. |
|
|
410
411
|
|
|
411
412
|
### What the screen looks like as text
|
|
@@ -585,6 +586,20 @@ simframe screens # what this device has learned
|
|
|
585
586
|
simframe goto invoices # walk there, verifying every step
|
|
586
587
|
```
|
|
587
588
|
|
|
589
|
+
And when the screen and the behaviour disagree, the question is usually not
|
|
590
|
+
about the screen at all:
|
|
591
|
+
|
|
592
|
+
```bash
|
|
593
|
+
simframe storage # apps with a data container
|
|
594
|
+
simframe storage com.example.myapp # what that app saved
|
|
595
|
+
```
|
|
596
|
+
|
|
597
|
+
`sim_ui` says what is drawn; `sim_storage` says what the app believes. It reads
|
|
598
|
+
the data container straight off the host filesystem, which means it works on a
|
|
599
|
+
device that is **shut down** — `simctl` cannot do this at all, on any of its own
|
|
600
|
+
paths, once a device stops running.
|
|
601
|
+
|
|
602
|
+
|
|
588
603
|
Measured on a four-tab tour, `goto` plans and walks three-step routes with every
|
|
589
604
|
step verified and no model call. It fails rather than guesses: an unknown
|
|
590
605
|
destination, a query matching two screens equally, or no path of known edges all
|
|
@@ -13,6 +13,10 @@ public enum Motion {
|
|
|
13
13
|
|
|
14
14
|
/// Below this mean absolute difference, two frames are the same picture.
|
|
15
15
|
public static let stillThreshold = 0.004
|
|
16
|
+
|
|
17
|
+
/// How many moving cells make an animation rather than sensor noise.
|
|
18
|
+
/// Set from the measurements in `state(history:now:)` below.
|
|
19
|
+
public static let minAnimatingCells = 12
|
|
16
20
|
/// Frames that must agree before the screen counts as settled.
|
|
17
21
|
///
|
|
18
22
|
/// Paired with a duration, because frame count alone is not a measure of
|
|
@@ -170,8 +174,35 @@ public enum Motion {
|
|
|
170
174
|
let cellThreshold = 0.06
|
|
171
175
|
let moving = union.filter { $0 > cellThreshold }.count
|
|
172
176
|
let fraction = Double(moving) / Double(union.count)
|
|
173
|
-
// Small and persistent, rather than a screen changing
|
|
174
|
-
|
|
177
|
+
// Small and persistent, rather than a screen changing — and big
|
|
178
|
+
// enough to be something.
|
|
179
|
+
//
|
|
180
|
+
// `fraction > 0` meant one cell of 4,608 counted as an animation,
|
|
181
|
+
// and measured on a device that fires constantly on screens where
|
|
182
|
+
// nothing is happening. A field report put it exactly right: a
|
|
183
|
+
// warning that is usually wrong trains the reader to ignore the one
|
|
184
|
+
// that matters.
|
|
185
|
+
//
|
|
186
|
+
// Measured on this device, 2026-09-14, which is what
|
|
187
|
+
// `minAnimatingCells` is set from:
|
|
188
|
+
//
|
|
189
|
+
// static home screen, nothing moving 1-4 cells, on 54% of frames
|
|
190
|
+
// a blinking text caret **2 cells** (1x2), on 100%
|
|
191
|
+
// the testbed's spinner 169 cells (13x13)
|
|
192
|
+
// Maps launching, median 656 cells (41x16)
|
|
193
|
+
// a real spinner, from the field 1,364 cells (44x31)
|
|
194
|
+
//
|
|
195
|
+
// Caret and noise sit at 1-4; the smallest real animation seen is
|
|
196
|
+
// 169. A 42x gap with nothing in it, so the threshold is not
|
|
197
|
+
// delicate — 12 is three times the worst noise and an order of
|
|
198
|
+
// magnitude below the weakest signal.
|
|
199
|
+
//
|
|
200
|
+
// The caret is the case this is really for, and it is why the
|
|
201
|
+
// question could not be settled by reasoning: a caret must never
|
|
202
|
+
// stop a screen from settling, and before this it did — 72 frames
|
|
203
|
+
// out of 72 with `settled: false` on a screen holding nothing but a
|
|
204
|
+
// text cursor.
|
|
205
|
+
if moving >= Self.minAnimatingCells && fraction < 0.06 {
|
|
175
206
|
animating = boundingBox(of: union, threshold: cellThreshold)
|
|
176
207
|
}
|
|
177
208
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "simframe",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.14.0",
|
|
4
4
|
"mcpName": "io.github.lvlrSajjad/simframe",
|
|
5
5
|
"description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
|
|
6
6
|
"keywords": [
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Run a CI step, and tell a sick simulator apart from a failing check.
|
|
3
|
+
//
|
|
4
|
+
// node scripts/ci-device-guard.mjs <udid> -- <command> [args...]
|
|
5
|
+
//
|
|
6
|
+
// **Why this exists, with the number that justifies it.** Over the last 25 CI
|
|
7
|
+
// runs the `integration` job failed 17 times and passed 5. Eight of the nine
|
|
8
|
+
// most recent failures name a device-state condition in their own output —
|
|
9
|
+
// `NSPOSIXErrorDomain code=60`, `the display produced no frame in 60s`, `the
|
|
10
|
+
// second app never launched` — and one of those runs was a **docs-only commit**
|
|
11
|
+
// that changed a single markdown file. Two runs of byte-identical code failed at
|
|
12
|
+
// two different steps.
|
|
13
|
+
//
|
|
14
|
+
// So the job has been answering two questions at once — *does simframe work* and
|
|
15
|
+
// *did this hosted simulator survive twenty minutes* — and the second dominates.
|
|
16
|
+
// A red build that is usually the device is the cry-wolf failure this repo keeps
|
|
17
|
+
// writing items about, and it cost two days of re-reading logs to learn nothing.
|
|
18
|
+
//
|
|
19
|
+
// This does not paper over failures. It distinguishes them: a named device
|
|
20
|
+
// condition gets the cure this project already ships (`simframe revive`) and one
|
|
21
|
+
// retry, exactly as DEFERRED 126 says to; anything else fails on the spot,
|
|
22
|
+
// untouched. The classification is written to the job summary so the *rate*
|
|
23
|
+
// becomes visible instead of arguable.
|
|
24
|
+
import { spawn } from 'node:child_process';
|
|
25
|
+
import fs from 'node:fs';
|
|
26
|
+
|
|
27
|
+
/** Conditions that are the simulator, not the code. Each seen in a real run. */
|
|
28
|
+
const DEVICE_STATE = [
|
|
29
|
+
[/NSPOSIXErrorDomain.*code=?\s*60|Operation timed out/i, 'simctl stopped answering (NSPOSIXErrorDomain 60)'],
|
|
30
|
+
[/did not produce a frame|produced no frame in \d+s/i, 'the daemon is up and the display renders nothing'],
|
|
31
|
+
[/Timeout waiting for screen surfaces|display surface is not answering|display surface could not be read/i, 'the display surface is wedged'],
|
|
32
|
+
[/no frames buffered|capture is wedged/i, 'capture stopped'],
|
|
33
|
+
[/the second app never launched|could not be dispatched/i, 'an app would not launch'],
|
|
34
|
+
];
|
|
35
|
+
|
|
36
|
+
const udid = process.argv[2];
|
|
37
|
+
const sep = process.argv.indexOf('--');
|
|
38
|
+
if (!udid || sep < 0) {
|
|
39
|
+
console.error('usage: ci-device-guard.mjs <udid> -- <command> [args...]');
|
|
40
|
+
process.exit(2);
|
|
41
|
+
}
|
|
42
|
+
const cmd = process.argv.slice(sep + 1);
|
|
43
|
+
|
|
44
|
+
function run(argv, { capture = true } = {}) {
|
|
45
|
+
return new Promise((resolve) => {
|
|
46
|
+
const p = spawn(argv[0], argv.slice(1), { stdio: capture ? ['inherit', 'pipe', 'pipe'] : 'inherit' });
|
|
47
|
+
let out = '';
|
|
48
|
+
p.stdout?.on('data', (d) => { out += d; process.stdout.write(d); });
|
|
49
|
+
p.stderr?.on('data', (d) => { out += d; process.stderr.write(d); });
|
|
50
|
+
p.on('close', (code) => resolve({ code, out }));
|
|
51
|
+
});
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const summary = (line) => {
|
|
55
|
+
const f = process.env.GITHUB_STEP_SUMMARY;
|
|
56
|
+
if (f) { try { fs.appendFileSync(f, `${line}\n`); } catch { /* summaries are a nicety */ } }
|
|
57
|
+
};
|
|
58
|
+
|
|
59
|
+
const deviceCause = (text) => DEVICE_STATE.find(([re]) => re.test(text))?.[1] ?? null;
|
|
60
|
+
|
|
61
|
+
const first = await run(cmd);
|
|
62
|
+
if (first.code === 0) process.exit(0);
|
|
63
|
+
|
|
64
|
+
const cause = deviceCause(first.out);
|
|
65
|
+
if (!cause) {
|
|
66
|
+
console.error(`\n (this step failed on its merits, not on the device — not retrying)`);
|
|
67
|
+
summary(`- \`${cmd.join(' ')}\` — **check failed** (exit ${first.code})`);
|
|
68
|
+
process.exit(first.code ?? 1);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
console.error(`\n (${cause} — DEFERRED 126. Reviving once and running again.)`);
|
|
72
|
+
await run(['node', 'src/cli.js', 'revive', `--device=${udid}`], { capture: false });
|
|
73
|
+
const second = await run(cmd);
|
|
74
|
+
if (second.code === 0) {
|
|
75
|
+
summary(`- \`${cmd.join(' ')}\` — passed after one revive (${cause})`);
|
|
76
|
+
process.exit(0);
|
|
77
|
+
}
|
|
78
|
+
// Twice in a row, on a condition we know the cure for. Reported as what it is.
|
|
79
|
+
const again = deviceCause(second.out);
|
|
80
|
+
summary(`- \`${cmd.join(' ')}\` — **${again ? 'device unavailable' : 'check failed'}** after a revive${again ? ` (${again})` : ''}`);
|
|
81
|
+
if (again) console.error(`\nFAIL the simulator is still in a bad state after a revive: ${again}`);
|
|
82
|
+
process.exit(second.code ?? 1);
|
|
@@ -67,14 +67,17 @@ read_state() {
|
|
|
67
67
|
printf '[{"button":"home"},{"settle":true}]\n' > /tmp/reset-local.json
|
|
68
68
|
node src/cli.js do /tmp/reset-local.json --device="$DEVICE" >/dev/null 2>&1
|
|
69
69
|
BEFORE=$(read_state)
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
70
|
+
# From inside an app, pressing home always changes the screen — see the job's
|
|
71
|
+
# comment for the four vehicles that did not hold. Setup gets to a known screen;
|
|
72
|
+
# the asserted action is simframe's own HID path with no simctl in it.
|
|
73
|
+
printf '[{"launch":{"value":"com.apple.Preferences","relaunch":true}},{"settle":true}]\n' > /tmp/setup-local.json
|
|
74
|
+
node src/cli.js do /tmp/setup-local.json --device="$DEVICE" >/tmp/step-local.log 2>&1 || true
|
|
75
|
+
BEFORE=$(read_state)
|
|
76
|
+
printf '[{"button":"home"},{"settle":true}]\n' > /tmp/flow-local.json
|
|
77
|
+
node src/cli.js do /tmp/flow-local.json --device="$DEVICE" >>/tmp/step-local.log 2>&1 || true
|
|
78
|
+
AFTER=$(read_state)
|
|
79
|
+
if [ "${BEFORE%% *}" != "${AFTER%% *}" ]; then ok "frame hash changed: ${BEFORE%% *} -> ${AFTER%% *}"
|
|
80
|
+
else bad "a step ran and capture saw no change"; tail -5 /tmp/step-local.log; fi
|
|
78
81
|
|
|
79
82
|
step "The memory layer — screen map, refs, graph, verdicts, flows"
|
|
80
83
|
# Mirrors the job: exit 75 means the display wedged and nothing was tested, so
|
package/scripts/ci-memory.mjs
CHANGED
|
@@ -111,7 +111,14 @@ async function cli(args, { expectFail = false, allowFail = false } = {}) {
|
|
|
111
111
|
// detail has been truncated for legibility, and the first version of this
|
|
112
112
|
// guard looked for "did not produce a frame" in a string that had been cut
|
|
113
113
|
// to "simframe daemon di". The full text only exists at this boundary.
|
|
114
|
-
|
|
114
|
+
//
|
|
115
|
+
// And when the payload is a JSON report, say what failed rather than
|
|
116
|
+
// handing back its first hundred characters. A run of this printed
|
|
117
|
+
// `simframe doctor --json failed: {\n "ok": false,\n "strict": true,\n
|
|
118
|
+
// "failu` — the word "failures" cut in half, one character before the only
|
|
119
|
+
// content that mattered. A harness that truncates away the reason is doing
|
|
120
|
+
// to its reader exactly what this repo keeps writing items about.
|
|
121
|
+
throw new Error(`simframe ${full.join(' ')} failed: ${summarise(why)}`);
|
|
115
122
|
}
|
|
116
123
|
}
|
|
117
124
|
|
|
@@ -193,6 +200,23 @@ async function jsonRetry(args, opts, attempts = 3) {
|
|
|
193
200
|
throw last;
|
|
194
201
|
}
|
|
195
202
|
|
|
203
|
+
/** A failed JSON report, reduced to the part that says what went wrong. */
|
|
204
|
+
function summarise(why) {
|
|
205
|
+
try {
|
|
206
|
+
const parsed = JSON.parse(why);
|
|
207
|
+
const failures = parsed.failures ?? parsed.failing ?? null;
|
|
208
|
+
if (Array.isArray(failures) && failures.length) {
|
|
209
|
+
return failures
|
|
210
|
+
.map((f) => (typeof f === 'string' ? f : `${f.name ?? f.check ?? '?'}: ${f.detail ?? f.note ?? f.message ?? ''}`.trim()))
|
|
211
|
+
.join('; ')
|
|
212
|
+
.slice(0, 400);
|
|
213
|
+
}
|
|
214
|
+
const bad = (parsed.checks ?? []).filter((c) => c.ok === false);
|
|
215
|
+
if (bad.length) return bad.map((c) => `${c.name}: ${c.detail ?? ''}`.trim()).join('; ').slice(0, 400);
|
|
216
|
+
} catch { /* not JSON, or not a shape we know — fall through to the raw text */ }
|
|
217
|
+
return why.slice(0, 400);
|
|
218
|
+
}
|
|
219
|
+
|
|
196
220
|
const markHash = async () => (await jsonRetry(['mark'])).hash;
|
|
197
221
|
|
|
198
222
|
function writeFlow(name, steps) {
|
|
@@ -86,6 +86,17 @@ console.log(`tour: ${tour.length} screens x ${rounds} rounds, every reading cold
|
|
|
86
86
|
const readings = [];
|
|
87
87
|
/** Navigations that did not land before the reading was taken. */
|
|
88
88
|
const arrivalFailures = [];
|
|
89
|
+
/** Readings too bare to be a screen — see the guard where this is used. */
|
|
90
|
+
const sparseReadings = [];
|
|
91
|
+
/**
|
|
92
|
+
* Below this, a reading cannot distinguish its screen from any other bare one.
|
|
93
|
+
*
|
|
94
|
+
* Not a failure threshold — see where it is used. The Settings root legitimately
|
|
95
|
+
* reads 4 tokens on a hosted runner, so treating this as "the screen has not
|
|
96
|
+
* drawn" rejected a real screen and made CI deterministically red. It marks
|
|
97
|
+
* readings worth suspecting when a score disagrees with itself, nothing more.
|
|
98
|
+
*/
|
|
99
|
+
const MIN_TOKENS_FOR_A_READING = 5;
|
|
89
100
|
|
|
90
101
|
/**
|
|
91
102
|
* The tokens that carry a name, as opposed to a shape.
|
|
@@ -124,6 +135,22 @@ const save = (extra = {}) => {
|
|
|
124
135
|
}, null, 2));
|
|
125
136
|
};
|
|
126
137
|
|
|
138
|
+
/**
|
|
139
|
+
* How hard to try for a frame newer than the navigation before giving up.
|
|
140
|
+
*
|
|
141
|
+
* Three re-reads a second apart is ~3s of slack against capture medians that
|
|
142
|
+
* were measured above 1s on a bad runner. Generous enough to absorb the spikes
|
|
143
|
+
* that caused this, small enough that a genuinely stopped capture still fails
|
|
144
|
+
* rather than hanging the job.
|
|
145
|
+
*/
|
|
146
|
+
const STALE_READ_RETRIES = 3;
|
|
147
|
+
const STALE_READ_WAIT_MS = 1000;
|
|
148
|
+
|
|
149
|
+
/** When the steps that were supposed to change the screen finished. */
|
|
150
|
+
let navigatedAt = 0;
|
|
151
|
+
/** Readings that never got a frame newer than their own navigation. */
|
|
152
|
+
const staleReadings = [];
|
|
153
|
+
|
|
127
154
|
for (let round = 1; round <= rounds; round += 1) {
|
|
128
155
|
for (const screen of tour) {
|
|
129
156
|
if (screen.steps?.length) {
|
|
@@ -140,6 +167,7 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
140
167
|
// matters.
|
|
141
168
|
try {
|
|
142
169
|
await actions.runScript(device, { steps: screen.steps, verify: false });
|
|
170
|
+
navigatedAt = Date.now();
|
|
143
171
|
} catch (err) {
|
|
144
172
|
save({ abandonedAt: { screen: screen.name, round, error: err.message } });
|
|
145
173
|
console.error(`\nFAIL round ${round}, "${screen.name}" never arrived: ${err.message}`);
|
|
@@ -148,7 +176,66 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
148
176
|
process.exit(1);
|
|
149
177
|
}
|
|
150
178
|
}
|
|
151
|
-
|
|
179
|
+
let id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
180
|
+
// A reading off a frame older than the navigation is not a reading either.
|
|
181
|
+
//
|
|
182
|
+
// Same rule as the sparseness guard below, on the other axis, and it took a
|
|
183
|
+
// third symptom to see they were one cause. Three CI runs failed this step
|
|
184
|
+
// three different ways — `settings` reading 4 tokens, a reading that "does
|
|
185
|
+
// not resemble its own screen", and a reading "taken on the previous
|
|
186
|
+
// screen" at similarity 1.00 off a frame **7168ms old**. All three are the
|
|
187
|
+
// same sentence: the reading is not of the screen we think it is, because
|
|
188
|
+
// capture on a hosted runner is slow. The daemon log for that run shows
|
|
189
|
+
// capture medians of 417ms, 503ms, 996ms, 1106ms and 1132ms against the
|
|
190
|
+
// 55.4ms p50 measured for a healthy runner — 20x, and with damage-driven
|
|
191
|
+
// capture a `fresh` read returns the newest frame that EXISTS, which on a
|
|
192
|
+
// runner that far behind can predate the navigation entirely.
|
|
193
|
+
//
|
|
194
|
+
// So the frame must have been captured after the steps that were supposed
|
|
195
|
+
// to change the screen. Re-read rather than fail, and fail only if it stays
|
|
196
|
+
// stale — an eval that scores a stale frame is measuring the runner, which
|
|
197
|
+
// is the one thing this harness says it is not doing.
|
|
198
|
+
if (navigatedAt) {
|
|
199
|
+
for (let attempt = 0; attempt < STALE_READ_RETRIES; attempt += 1) {
|
|
200
|
+
const capturedAt = id.state?.capturedAt ?? 0;
|
|
201
|
+
if (capturedAt >= navigatedAt) break;
|
|
202
|
+
const age = Date.now() - capturedAt;
|
|
203
|
+
console.log(` ("${screen.name}" read a frame from ${age}ms ago, older than the navigation`
|
|
204
|
+
+ ` — reading again ${attempt + 1}/${STALE_READ_RETRIES})`);
|
|
205
|
+
await new Promise((r) => setTimeout(r, STALE_READ_WAIT_MS));
|
|
206
|
+
id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
207
|
+
}
|
|
208
|
+
if ((id.state?.capturedAt ?? 0) < navigatedAt) {
|
|
209
|
+
staleReadings.push(`${screen.name} round ${round}: frame still predates the navigation`
|
|
210
|
+
+ ` by ${navigatedAt - (id.state?.capturedAt ?? 0)}ms after ${STALE_READ_RETRIES} re-reads`);
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
// A reading too sparse to be a screen is not a reading.
|
|
214
|
+
//
|
|
215
|
+
// On a runner measured at ~2.5x slower than a laptop (EXPERIMENTS §15),
|
|
216
|
+
// perception can land on a half-rendered screen: two CI runs recorded
|
|
217
|
+
// `settings` and `settings-general` **both at 4 tokens**, identical hash,
|
|
218
|
+
// and the eval duly reported that a reading did not resemble its own
|
|
219
|
+
// screen. It resembled nothing, because almost nothing had been drawn yet.
|
|
220
|
+
//
|
|
221
|
+
// This is the hazard `TOKEN_RULES_VERSION` 7 was written for, arriving
|
|
222
|
+
// through the harness instead of the rules: *"two sparse nameless readings
|
|
223
|
+
// then matched exactly, one hash standing for two different screens"*. A
|
|
224
|
+
// guard exists for identity and there was none here.
|
|
225
|
+
//
|
|
226
|
+
// Read again rather than fail, and fail only if it stays sparse — the same
|
|
227
|
+
// "untested is not passed" rule the memory harness learned. A screen that
|
|
228
|
+
// is genuinely this bare after a second look is a real finding.
|
|
229
|
+
if ((id.tokens ?? []).length < MIN_TOKENS_FOR_A_READING) {
|
|
230
|
+
const before = (id.tokens ?? []).length;
|
|
231
|
+
await new Promise((r) => setTimeout(r, 1500));
|
|
232
|
+
id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
233
|
+
const after = (id.tokens ?? []).length;
|
|
234
|
+
console.log(` ("${screen.name}" read ${before} token(s) — too sparse to compare; read again: ${after})`);
|
|
235
|
+
if (after < MIN_TOKENS_FOR_A_READING) {
|
|
236
|
+
sparseReadings.push(`${screen.name} round ${round}: ${after} token(s) after two reads`);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
152
239
|
// Did we actually arrive? Two differently-named screens reading the same
|
|
153
240
|
// fingerprint means the navigation did not land before the reading was
|
|
154
241
|
// taken, and every distribution below it is then measuring the tour rather
|
|
@@ -219,9 +306,45 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
219
306
|
}
|
|
220
307
|
}
|
|
221
308
|
|
|
222
|
-
save();
|
|
309
|
+
save({ staleReadings });
|
|
223
310
|
if (outFile) console.log(`\nwrote ${readings.length} readings to ${outFile}`);
|
|
224
311
|
|
|
312
|
+
if (sparseReadings.length) {
|
|
313
|
+
// Reported, NOT failed — and that correction is worth more than the check.
|
|
314
|
+
//
|
|
315
|
+
// This exited 1 for four hours on 2026-09-14 and turned an intermittent CI
|
|
316
|
+
// failure into a deterministic one. The reasoning was that a 4-token reading
|
|
317
|
+
// is a screen that has not drawn. It is not: the Settings root legitimately
|
|
318
|
+
// reads 4 tokens on a hosted runner, twice in a row, three times in a row —
|
|
319
|
+
// which the comment on `MIN_TOKENS_FOR_A_READING` had itself said ("4-8
|
|
320
|
+
// tokens") one screen above the code that rejected it.
|
|
321
|
+
//
|
|
322
|
+
// What remains true is that a reading this bare cannot distinguish its screen
|
|
323
|
+
// from anything else equally bare, so it is worth saying out loud. What is not
|
|
324
|
+
// true is that saying so should stop the run. The arrival check below is the
|
|
325
|
+
// one that catches the real collision, and it does so on evidence rather than
|
|
326
|
+
// on a token count.
|
|
327
|
+
console.log(`\nNOTE ${sparseReadings.length} reading(s) were too bare to distinguish a screen:`);
|
|
328
|
+
for (const f of sparseReadings) console.log(` ${f}`);
|
|
329
|
+
console.log(`\nFewer than ${MIN_TOKENS_FOR_A_READING} tokens after two reads. That is not automatically`);
|
|
330
|
+
console.log('wrong — a plain screen really can be this bare — but two readings this sparse');
|
|
331
|
+
console.log('cannot be told apart, which is the hazard TOKEN_RULES_VERSION 7 was written');
|
|
332
|
+
console.log('for. If a same-screen score below disagrees with itself, start here.');
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
if (staleReadings.length) {
|
|
336
|
+
// A NOTE and not a failure, for the same reason the sparseness guard is one:
|
|
337
|
+
// this says the runner was too slow to give a fresh frame, which is a fact
|
|
338
|
+
// about the machine and not about the fingerprint. It is printed *above* the
|
|
339
|
+
// arrival check on purpose — when both fire, this is the explanation of that.
|
|
340
|
+
console.log(`\nNOTE ${staleReadings.length} reading(s) never got a frame newer than their navigation:`);
|
|
341
|
+
for (const f of staleReadings) console.log(` ${f}`);
|
|
342
|
+
console.log(`\nCapture on this host is behind the tour. With damage-driven capture a "fresh"`);
|
|
343
|
+
console.log('read returns the newest frame that exists, so on a slow runner it can predate');
|
|
344
|
+
console.log('the navigation entirely. If an arrival failure follows, this is its cause and');
|
|
345
|
+
console.log('the tour is not what needs fixing.');
|
|
346
|
+
}
|
|
347
|
+
|
|
225
348
|
if (arrivalFailures.length) {
|
|
226
349
|
console.error(`\nFAIL ${arrivalFailures.length} reading(s) were taken on the previous screen:`);
|
|
227
350
|
for (const f of arrivalFailures) console.error(` ${f}`);
|
package/src/actions.js
CHANGED
|
@@ -2555,7 +2555,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2555
2555
|
for (let attempt = 0; ; attempt += 1) {
|
|
2556
2556
|
for (const one of alternatives) {
|
|
2557
2557
|
try {
|
|
2558
|
-
const found = await api.locate(deviceQuery, one, { refresh:
|
|
2558
|
+
const found = await api.locate(deviceQuery, one, { refresh: step.refresh !== false, options: ctx.options });
|
|
2559
2559
|
return `${JSON.stringify(one)} appeared at ${found.target.x},${found.target.y}`
|
|
2560
2560
|
+ ` (first of ${alternatives.length} awaited)`;
|
|
2561
2561
|
} catch (err) {
|
|
@@ -2563,10 +2563,12 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2563
2563
|
}
|
|
2564
2564
|
}
|
|
2565
2565
|
if (Date.now() >= limit) {
|
|
2566
|
+
const stillAny = await stillnessNote(deviceQuery, ctx);
|
|
2566
2567
|
throw new Error(
|
|
2567
2568
|
`none of ${alternatives.length} awaited strings appeared`
|
|
2568
2569
|
+ ` (${alternatives.map((a) => JSON.stringify(a)).join(', ')}) in ${step.timeoutMs ?? 8000}ms.`
|
|
2569
|
-
+ ` Last: ${lastError}
|
|
2570
|
+
+ ` Last: ${lastError}`
|
|
2571
|
+
+ (stillAny ? ` (${stillAny.note})` : ''),
|
|
2570
2572
|
);
|
|
2571
2573
|
}
|
|
2572
2574
|
await api.waitFor(deviceQuery, { mode: 'stable', stableMs: 200, timeoutMs: 700, options: ctx.options })
|
|
@@ -2575,7 +2577,26 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2575
2577
|
}
|
|
2576
2578
|
for (let attempt = 0; ; attempt += 1) {
|
|
2577
2579
|
try {
|
|
2578
|
-
|
|
2580
|
+
// Fresh, like `assert` — and for the same reason, found the same way.
|
|
2581
|
+
//
|
|
2582
|
+
// This read `refresh: attempt > 0`, so the FIRST look resolved against
|
|
2583
|
+
// the remembered map. A wait that can be satisfied by memory is not a
|
|
2584
|
+
// wait: if recall hands back the wrong screen's map — which it does
|
|
2585
|
+
// when two screens collide on a layout hash — the target "appears"
|
|
2586
|
+
// without ever having been on screen, instantly, and the caller
|
|
2587
|
+
// proceeds against a screen it is not on.
|
|
2588
|
+
//
|
|
2589
|
+
// Caught by the fingerprint eval on CI, which is the only place it
|
|
2590
|
+
// could show: `settings-general` read the **Settings root** and its
|
|
2591
|
+
// `waitFor "About"` had passed. The signature is in the round numbers
|
|
2592
|
+
// — r2 and r3, never r1 — because memory has to be warm before it can
|
|
2593
|
+
// lie, and round 1 is always cold.
|
|
2594
|
+
//
|
|
2595
|
+
// `assert` was made fresh by default after this exact defect cost a
|
|
2596
|
+
// reported session; `waitFor` was left as it was. One perception pass
|
|
2597
|
+
// is the price, and it is the same trade: a read is cheaper than the
|
|
2598
|
+
// round trip a wrong verdict causes.
|
|
2599
|
+
const found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh !== false, options: ctx.options });
|
|
2579
2600
|
return `"${found.target.label}" appeared at ${found.target.x},${found.target.y}`;
|
|
2580
2601
|
} catch (err) {
|
|
2581
2602
|
lastError = err.message;
|
|
@@ -2609,9 +2630,23 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2609
2630
|
}
|
|
2610
2631
|
}
|
|
2611
2632
|
if (Date.now() >= limit) break;
|
|
2633
|
+
// Opt-in: stop early on a screen that has plainly stopped changing.
|
|
2634
|
+
if (Number.isFinite(step.failIfStillFor)) {
|
|
2635
|
+
const still = await stillnessNote(deviceQuery, ctx);
|
|
2636
|
+
if (still && still.ms >= step.failIfStillFor) {
|
|
2637
|
+
throw new Error(`gave up on ${query} after ${Date.now() - (limit - (step.timeoutMs ?? 8000))}ms:`
|
|
2638
|
+
+ ` ${still.note}, which is past the ${step.failIfStillFor}ms you said to stop at.`
|
|
2639
|
+
+ ` Last: ${lastError}`);
|
|
2640
|
+
}
|
|
2641
|
+
}
|
|
2612
2642
|
await sleep(POLL_MS);
|
|
2613
2643
|
}
|
|
2614
|
-
|
|
2644
|
+
// Say how long it had been still. A wait that burned three minutes on a
|
|
2645
|
+
// screen static for the last twelve seconds should not make the reader
|
|
2646
|
+
// work that out from a second command.
|
|
2647
|
+
const still = await stillnessNote(deviceQuery, ctx);
|
|
2648
|
+
throw new Error(`waited ${step.timeoutMs ?? 8000}ms for ${query}: ${lastError}`
|
|
2649
|
+
+ (still ? ` (${still.note} — pass failIfStillFor to stop early next time)` : ''));
|
|
2615
2650
|
}
|
|
2616
2651
|
|
|
2617
2652
|
// One assert step for every condition, because `assertText` could only ask
|
|
@@ -2639,6 +2674,32 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2639
2674
|
// is the opposite of what you want learned."*
|
|
2640
2675
|
found = await api.locate(deviceQuery, query, { index: step.index, refresh: step.refresh !== false, options: ctx.options });
|
|
2641
2676
|
} catch (err) {
|
|
2677
|
+
// Ambiguity answers "is it there", and it answers it *yes*.
|
|
2678
|
+
//
|
|
2679
|
+
// `locate` refuses a query that matches several elements, which is right
|
|
2680
|
+
// for `tap` — picking the wrong one of two taps the wrong thing — and
|
|
2681
|
+
// wrong for a question about presence. Reported from the field on
|
|
2682
|
+
// 0.13.0: `assert Administrator is visible` failed, and aborted the rest
|
|
2683
|
+
// of the batch, against a screen with **two** Administrators on it. The
|
|
2684
|
+
// assertion's own semantics were satisfied twice over. The message was
|
|
2685
|
+
// praised in the same breath — candidates, coordinates and scores — so
|
|
2686
|
+
// what was wrong was the verdict, not the diagnosis.
|
|
2687
|
+
//
|
|
2688
|
+
// The `gone` direction had the mirror-image bug and nobody had hit it
|
|
2689
|
+
// yet: any error at all returned "is gone", so a query matching two
|
|
2690
|
+
// visible elements would have reported them absent. Ambiguity is the one
|
|
2691
|
+
// error that is positive evidence of presence, and it was being read as
|
|
2692
|
+
// proof of absence.
|
|
2693
|
+
//
|
|
2694
|
+
// Strict single-match resolution stays for `enabled`, `disabled` and
|
|
2695
|
+
// `value`, where the question is about a *particular* element and
|
|
2696
|
+
// answering it from the wrong one is how a confident wrong answer gets
|
|
2697
|
+
// made.
|
|
2698
|
+
if (err.ambiguous && err.candidates?.length) {
|
|
2699
|
+
const n = err.candidates.length;
|
|
2700
|
+
if (want === 'visible') return `${query} is visible (${n} things match it here — a presence check does not have to choose)`;
|
|
2701
|
+
if (want === 'gone') throw new Error(`${query}: still here — ${n} things on this screen match it`);
|
|
2702
|
+
}
|
|
2642
2703
|
if (want === 'gone') return `${query} is gone`;
|
|
2643
2704
|
throw new Error(`${query}: ${err.message}`);
|
|
2644
2705
|
}
|
|
@@ -2744,3 +2805,55 @@ export function settleEvidence(w) {
|
|
|
2744
2805
|
return (parts.length ? parts.join('; ') : 'nothing observed')
|
|
2745
2806
|
+ (m ? `\n${m.map}` : '');
|
|
2746
2807
|
}
|
|
2808
|
+
|
|
2809
|
+
/**
|
|
2810
|
+
* The one-line flow summary, so `ok` and `FAIL` each mean exactly one thing.
|
|
2811
|
+
*
|
|
2812
|
+
* `FLOW FAILED — 3/3 steps` was reported from the field as reading like a
|
|
2813
|
+
* success, and the reporter had it exactly: the denominator means *attempted*
|
|
2814
|
+
* on failure and *succeeded* on success, so the same shape carries opposite
|
|
2815
|
+
* meanings. `flow completed — 5/5 steps` and `FLOW FAILED — 3/3 steps` differ
|
|
2816
|
+
* only in a word, and the numbers argue against the word.
|
|
2817
|
+
*
|
|
2818
|
+
* On failure it says how many worked and how many did not, which is the thing a
|
|
2819
|
+
* caller has to know to decide whether to resume or re-plan.
|
|
2820
|
+
*/
|
|
2821
|
+
export function flowSummary(res, { withTime = true } = {}) {
|
|
2822
|
+
const time = withTime && Number.isFinite(res.totalMs) ? ` in ${res.totalMs}ms` : '';
|
|
2823
|
+
if (res.ok) return `flow completed — ${res.ranSteps}/${res.totalSteps} steps${time}`;
|
|
2824
|
+
const failed = (res.results ?? []).filter((r) => r.ok === false).length || 1;
|
|
2825
|
+
const worked = Math.max(0, res.ranSteps - failed);
|
|
2826
|
+
const unattempted = Math.max(0, res.totalSteps - res.ranSteps);
|
|
2827
|
+
return `FLOW FAILED — ${worked} ok, ${failed} failed`
|
|
2828
|
+
+ (unattempted ? `, ${unattempted} not attempted` : '')
|
|
2829
|
+
+ ` (of ${res.totalSteps})${time}`;
|
|
2830
|
+
}
|
|
2831
|
+
|
|
2832
|
+
/**
|
|
2833
|
+
* How long the screen has been still, for a wait that is about to give up.
|
|
2834
|
+
*
|
|
2835
|
+
* A field report: a 180-second wait for a control that never appeared, on a
|
|
2836
|
+
* screen that had been static for about twelve of those seconds — the app had
|
|
2837
|
+
* logged itself out and was sitting on a login form. The failure message was
|
|
2838
|
+
* praised for listing what *was* on screen; it arrived three minutes late.
|
|
2839
|
+
*
|
|
2840
|
+
* Stillness is already tracked and already printed by other commands, so the
|
|
2841
|
+
* information existed and this wait simply never asked for it. Reported on
|
|
2842
|
+
* every timeout, and — only when the caller opts in — allowed to end the wait
|
|
2843
|
+
* early. Opt-in and not default, because a still screen is exactly what a
|
|
2844
|
+
* pending network call looks like: the target may yet arrive, and a wait that
|
|
2845
|
+
* gave up on stillness alone would break the case waits exist for.
|
|
2846
|
+
*
|
|
2847
|
+
* This is only trustworthy because of 134. Before the animation threshold was
|
|
2848
|
+
* measured, a completely static screen claimed something was animating on 54%
|
|
2849
|
+
* of its frames, and "nothing has moved" could not be said with a straight face.
|
|
2850
|
+
*/
|
|
2851
|
+
async function stillnessNote(deviceQuery, ctx) {
|
|
2852
|
+
try {
|
|
2853
|
+
const { state } = await api.getState(deviceQuery, { options: ctx.options });
|
|
2854
|
+
const ms = state?.stableForMs;
|
|
2855
|
+
return Number.isFinite(ms) ? { ms, note: `the screen has not moved for ${Math.round(ms)}ms` } : null;
|
|
2856
|
+
} catch {
|
|
2857
|
+
return null;
|
|
2858
|
+
}
|
|
2859
|
+
}
|
package/src/cli.js
CHANGED
|
@@ -12,6 +12,7 @@ import * as baseline from './baseline.js';
|
|
|
12
12
|
import * as metrics from './metrics.js';
|
|
13
13
|
import * as navigate from './navigate.js';
|
|
14
14
|
import { decodePng } from './png.js';
|
|
15
|
+
import * as storage from './storage.js';
|
|
15
16
|
import * as store from './store.js';
|
|
16
17
|
import * as view from './view.js';
|
|
17
18
|
|
|
@@ -33,6 +34,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
33
34
|
simframe tap <selector> tap #3, "Save", or @120,400
|
|
34
35
|
simframe do <script.json> run a scripted flow (see below)
|
|
35
36
|
simframe screens [device] list screens this device has learned
|
|
37
|
+
simframe storage [bundle-id] what the app saved (works on a shut-down device)
|
|
36
38
|
simframe goto <screen> walk to a known screen through known steps
|
|
37
39
|
simframe flow save <name> <script.json> run a flow and save it if every step verifies
|
|
38
40
|
simframe flow run <name> replay a saved flow
|
|
@@ -801,7 +803,7 @@ async function main() {
|
|
|
801
803
|
},
|
|
802
804
|
[
|
|
803
805
|
...res.results.map(stepLine),
|
|
804
|
-
|
|
806
|
+
actions.flowSummary(res),
|
|
805
807
|
saved && (saved.ok ? `saved flow "${saved.name}" — ${saved.steps} steps` : `not saved: ${saved.reason}`),
|
|
806
808
|
map && `\n${map}`,
|
|
807
809
|
],
|
|
@@ -860,6 +862,25 @@ async function main() {
|
|
|
860
862
|
return;
|
|
861
863
|
}
|
|
862
864
|
|
|
865
|
+
case 'storage': {
|
|
866
|
+
// A file read, like `screens`, and deliberately not a daemon call. The
|
|
867
|
+
// whole value of this command is that it answers on a device that is not
|
|
868
|
+
// running — measured: simctl itself cannot, on this Xcode. Routing it
|
|
869
|
+
// through the daemon would throw that away for no gain.
|
|
870
|
+
const device = await resolveDevice(flags.device);
|
|
871
|
+
const [bundleId] = positional;
|
|
872
|
+
if (!bundleId) {
|
|
873
|
+
const list = await storage.apps(device.udid);
|
|
874
|
+
const match = flags.match ? String(flags.match).toLowerCase() : null;
|
|
875
|
+
const shown = match ? list.filter((a) => a.bundleId.toLowerCase().includes(match)) : list;
|
|
876
|
+
emit(flags, shown, storage.formatApps(shown));
|
|
877
|
+
return;
|
|
878
|
+
}
|
|
879
|
+
const result = await storage.read(device.udid, bundleId);
|
|
880
|
+
emit(flags, result, storage.format(result));
|
|
881
|
+
return;
|
|
882
|
+
}
|
|
883
|
+
|
|
863
884
|
case 'screens': {
|
|
864
885
|
// Reading what this device has learned is a file read. It used to go
|
|
865
886
|
// through ensureDaemon, so a device whose capture had stopped could not
|
|
@@ -920,7 +941,7 @@ async function main() {
|
|
|
920
941
|
}
|
|
921
942
|
emit(flags, res, [
|
|
922
943
|
...(res.results ?? []).map(stepLine),
|
|
923
|
-
|
|
944
|
+
actions.flowSummary(res, { withTime: false }),
|
|
924
945
|
]);
|
|
925
946
|
process.exitCode = res.ok ? 0 : 1;
|
|
926
947
|
return;
|