simframe 0.14.0 → 0.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "simframe",
3
- "version": "0.14.0",
3
+ "version": "0.14.1",
4
4
  "mcpName": "io.github.lvlrSajjad/simframe",
5
5
  "description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
6
6
  "keywords": [
@@ -225,10 +225,28 @@ function writeFlow(name, steps) {
225
225
  return file;
226
226
  }
227
227
 
228
- // A closed loop: openUrl puts Safari in front, home leaves it. Both ends are
229
- // screens the graph can learn, and every pass starts where the last one ended.
228
+ // A closed loop: launching an app puts it in front, home leaves it. Both ends
229
+ // are screens the graph can learn, and every pass starts where the last one
230
+ // ended.
231
+ //
232
+ // **It used to be `openUrl https://example.com`, and that was the bug.** Item
233
+ // 142 catalogued `simctl openurl` timing out on a loaded runner as a failure
234
+ // class, marked it "fixed — vehicle changed", and changed the vehicle in the
235
+ // *workflow's* step only. This loop was left on it, and so was the novel action
236
+ // below. That is the same class-versus-symptom error the `waitFor`/`assert`
237
+ // twin recorded: the fix went where the report pointed instead of everywhere
238
+ // the cause reached.
239
+ //
240
+ // It came back on 2026-09-14: passes 2 and 3 halted at step 0, the graph never
241
+ // got the chance to predict, and the failure read as "the outcome is predicted
242
+ // — pass 0", which names the graph for something Safari did.
243
+ //
244
+ // The replacement is the vehicle item 142 measured and proved for exactly this:
245
+ // **from inside an app, pressing home always changes the screen.** Settings is
246
+ // already installed everywhere this runs, the launch needs no network, and
247
+ // neither end depends on a browser cold-starting on a shared machine.
230
248
  const LOOP = writeFlow('simframe-ci-loop.json', [
231
- { openUrl: 'https://example.com' },
249
+ { launch: { value: 'com.apple.Preferences', relaunch: true } },
232
250
  { button: 'home' },
233
251
  ]);
234
252
  // Leaving whatever screen the map was read on.
@@ -537,10 +555,23 @@ if (novelRan && novelMoved) {
537
555
  // so a run in which every pass failed to dispatch says nothing about
538
556
  // prediction. It failed the build as `pass 0` while the real cause was a
539
557
  // simctl launch timing out, three checks upstream.
540
- const anyPassRan = passes.some((p) => Array.isArray(p.run?.results) && p.run.results.some((r) => r.ok !== false));
541
- if (!anyPassRan) {
558
+ //
559
+ // **The rule was right and the test of it was too coarse.** `anyPassRan` asks
560
+ // whether *any* pass ran, but prediction can only be observed on a pass AFTER
561
+ // the one that taught the edge — so pass 1 running is not enough. On
562
+ // 2026-09-14 pass 1 ran, passes 2 and 3 halted at step 0, and this reported
563
+ // `pass 0` as though the graph had declined to predict. Nothing had asked it
564
+ // to. Same sentence as the novel action three checks above: untested is not
565
+ // broken, and the guard has to test the pass the claim actually depends on.
566
+ const dispatched = (p) => Array.isArray(p.run?.results) && p.run.results.some((r) => r.ok !== false);
567
+ const laterPassRan = passes.slice(1).some(dispatched);
568
+ if (!passes.some(dispatched)) {
542
569
  skip('and once the graph has seen it, the outcome is predicted',
543
570
  'no pass dispatched a step, so the graph was never given anything to learn');
571
+ } else if (!laterPassRan) {
572
+ skip('and once the graph has seen it, the outcome is predicted',
573
+ `only the first pass dispatched a step (${passes.slice(1).map((p, i) => `pass ${i + 2}: [${p.verdicts.join(', ')}]`).join('; ')})`
574
+ + ' — prediction is only observable on a pass after the one that taught the edge');
544
575
  } else {
545
576
  check(passes.some((p) => p.verdicts.includes('ok')),
546
577
  'and once the graph has seen it, the outcome is predicted',
@@ -88,6 +88,8 @@ const readings = [];
88
88
  const arrivalFailures = [];
89
89
  /** Readings too bare to be a screen — see the guard where this is used. */
90
90
  const sparseReadings = [];
91
+ /** `name|round` of every reading that stayed too sparse, so the report below can name the cause. */
92
+ const sparseAt = new Set();
91
93
  /**
92
94
  * Below this, a reading cannot distinguish its screen from any other bare one.
93
95
  *
@@ -144,6 +146,14 @@ const save = (extra = {}) => {
144
146
  * rather than hanging the job.
145
147
  */
146
148
  const STALE_READ_RETRIES = 3;
149
+ /**
150
+ * How hard to try for a reading rich enough to tell its screen apart.
151
+ *
152
+ * Backed off rather than fixed, because the thing being waited for is a screen
153
+ * finishing its draw, and the runner that needs this is the slow one.
154
+ */
155
+ const SPARSE_READ_RETRIES = 3;
156
+ const SPARSE_READ_WAIT_MS = 1500;
147
157
  const STALE_READ_WAIT_MS = 1000;
148
158
 
149
159
  /** When the steps that were supposed to change the screen finished. */
@@ -151,6 +161,52 @@ let navigatedAt = 0;
151
161
  /** Readings that never got a frame newer than their own navigation. */
152
162
  const staleReadings = [];
153
163
 
164
+ /**
165
+ * Visit every screen once before measuring, taking no readings.
166
+ *
167
+ * **The rounds are supposed to be repeat measurements of one thing.** They were
168
+ * not: round 1 systematically differed from rounds 2 and 3 whenever an app was
169
+ * cold, because an app that has just launched has not finished publishing its
170
+ * accessibility tree. Measured on CI — round 1 of `browser` read `e00725fce7`
171
+ * with **0 named** elements where rounds 2 and 3 read `9589eb2471` with **5**,
172
+ * while every other screen matched across all three. That is the eval measuring
173
+ * app cold-start, which it never set out to measure and does not report.
174
+ *
175
+ * It was invisible for a long time because something else was paying for it:
176
+ * the memory-layer step runs first on CI and drives the same apps for 504s,
177
+ * Safari included, so the eval always met a warm device. Sharding the job
178
+ * removed that neighbour and the dependency surfaced immediately. The defect
179
+ * was always here; the neighbour was hiding it.
180
+ *
181
+ * This is **not** the same as making the readings warm. Every reading is still
182
+ * taken with `fresh: true` against cold screen memory, which is what "cold"
183
+ * means in this harness — the graph must not have seen the screen before. What
184
+ * the warm-up removes is a variable about the *operating system* that the
185
+ * fingerprint has nothing to do with.
186
+ *
187
+ * Skippable with `--no-warmup`, because the comparison is the evidence: run it
188
+ * both ways to see whether round 1 still disagrees with its own repeats.
189
+ */
190
+ async function warmUp() {
191
+ const started = Date.now();
192
+ console.log('warming: visiting each screen once, taking no readings');
193
+ for (const screen of tour) {
194
+ if (!screen.steps?.length) continue;
195
+ try {
196
+ await actions.runScript(device, { steps: screen.steps, verify: false });
197
+ } catch (err) {
198
+ // A warm-up failure is not a result. The measured rounds below will meet
199
+ // the same screen and fail there with the harness's own reporting, which
200
+ // says which screen and writes the readings out. Failing here would cost
201
+ // that and report a screen that was never measured.
202
+ console.log(` (warm-up could not reach "${screen.name}": ${err.message})`);
203
+ }
204
+ }
205
+ console.log(`warming: done in ${Math.round((Date.now() - started) / 1000)}s\n`);
206
+ }
207
+
208
+ if (!process.argv.includes('--no-warmup')) await warmUp();
209
+
154
210
  for (let round = 1; round <= rounds; round += 1) {
155
211
  for (const screen of tour) {
156
212
  if (screen.steps?.length) {
@@ -176,7 +232,30 @@ for (let round = 1; round <= rounds; round += 1) {
176
232
  process.exit(1);
177
233
  }
178
234
  }
179
- let id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
235
+ // A read can throw, and when it does the readings so far are the evidence.
236
+ //
237
+ // This file already says so — *"Everything read so far is written out
238
+ // first, because a failing run is the one whose evidence matters"* — and
239
+ // then left this call unguarded, so a runner whose OCR overran its budget
240
+ // produced a raw stack trace, no `--out` file, and nothing for
241
+ // `analyse-fingerprint.mjs` to read. Exactly the failure the `save()` above
242
+ // was written to prevent, one line away from it.
243
+ //
244
+ // Not retried here. The read budget is already the client's own give-up
245
+ // point, so a read that overran it is a statement about the machine, and
246
+ // the honest thing is to say which screen and how far the tour got.
247
+ let id;
248
+ try {
249
+ id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
250
+ } catch (err) {
251
+ save({ abandonedAt: { screen: screen.name, round, error: err.message, phase: 'reading' } });
252
+ console.error(`\nFAIL round ${round}, "${screen.name}" could not be read: ${err.message}`);
253
+ console.error(`${readings.length} reading(s) taken before this were written out.`);
254
+ console.error('A read that overran its budget is a fact about the host, not about the');
255
+ console.error('fingerprint — the budget is already the client\'s own give-up point, so');
256
+ console.error('raising it would only move where the wait ends.');
257
+ process.exit(1);
258
+ }
180
259
  // A reading off a frame older than the navigation is not a reading either.
181
260
  //
182
261
  // Same rule as the sparseness guard below, on the other axis, and it took a
@@ -227,13 +306,36 @@ for (let round = 1; round <= rounds; round += 1) {
227
306
  // "untested is not passed" rule the memory harness learned. A screen that
228
307
  // is genuinely this bare after a second look is a real finding.
229
308
  if ((id.tokens ?? []).length < MIN_TOKENS_FOR_A_READING) {
309
+ // Read again, and keep the BEST reading rather than the last one.
310
+ //
311
+ // One retry was not enough and the failure it produced pointed at the
312
+ // wrong thing. On CI, `settings-general` read 4 tokens twice and then
313
+ // scored 1.00 against the *Settings root* — because the General screen's
314
+ // back button is labelled "Settings", so a General seen only down to its
315
+ // nav bar is structurally the same screen as the root. The harness
316
+ // reported it as a fingerprint that does not resemble its own screen. The
317
+ // fingerprint was fine; the look was too short. `settings-general` reads
318
+ // 5 tokens when it is read properly.
319
+ //
320
+ // Best-of, not last, because these reads are samples of a screen that is
321
+ // still drawing: a later read is usually richer but not reliably so, and
322
+ // throwing away a 5-token reading because the retry saw 4 would be the
323
+ // same bug with more steps.
230
324
  const before = (id.tokens ?? []).length;
231
- await new Promise((r) => setTimeout(r, 1500));
232
- id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
325
+ const seen = [before];
326
+ for (let attempt = 1; attempt <= SPARSE_READ_RETRIES; attempt += 1) {
327
+ await new Promise((r) => setTimeout(r, SPARSE_READ_WAIT_MS * attempt));
328
+ const again = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
329
+ seen.push((again.tokens ?? []).length);
330
+ if ((again.tokens ?? []).length > (id.tokens ?? []).length) id = again;
331
+ if ((id.tokens ?? []).length >= MIN_TOKENS_FOR_A_READING) break;
332
+ }
233
333
  const after = (id.tokens ?? []).length;
234
- console.log(` ("${screen.name}" read ${before} token(s) — too sparse to compare; read again: ${after})`);
334
+ console.log(` ("${screen.name}" read ${before} token(s) — too sparse to compare;`
335
+ + ` read again: ${seen.slice(1).join(', ')} — kept ${after})`);
235
336
  if (after < MIN_TOKENS_FOR_A_READING) {
236
- sparseReadings.push(`${screen.name} round ${round}: ${after} token(s) after two reads`);
337
+ sparseReadings.push(`${screen.name} round ${round}: ${after} token(s) after ${seen.length} reads`);
338
+ sparseAt.add(`${screen.name}|${round}`);
237
339
  }
238
340
  }
239
341
  // Did we actually arrive? Two differently-named screens reading the same
@@ -419,6 +521,20 @@ if (strays.length) {
419
521
  const matchNamed = match ? namedTokens(match.tokens) : [];
420
522
  const collided = bestOther >= 0.99 && named.length === 0 && matchNamed.length === 0;
421
523
  if (collided) collisions += 1;
524
+ // The third cause, and the one that produced this report on 2026-09-15.
525
+ //
526
+ // A reading that stayed under the token floor did not fail to resemble its
527
+ // screen — it never saw enough of the screen to resemble anything. On CI
528
+ // `settings-general` read 4 tokens and scored 1.00 against the Settings
529
+ // root, because the General screen's back button is labelled "Settings", so
530
+ // a General seen only down to its nav bar IS the root structurally. The
531
+ // report called that a wrong turn and sent the reader to fix the tour.
532
+ //
533
+ // Not folded into `collided` above: that one means the fingerprint had
534
+ // nothing to work with, which is this harness's subject. This means we did
535
+ // not look long enough, which is the harness's own fault and a different
536
+ // remedy.
537
+ const underRead = sparseAt.has(`${reading.name}|${reading.round}`);
422
538
  console.error(` ${reading.name} r${reading.round}: own screen ${bestSelf.toFixed(2)}, `
423
539
  + `${match ? `${match.name} r${match.round}` : 'another screen'} ${bestOther.toFixed(2)} `
424
540
  + `(${reading.count} tokens, ${named.length} named, sources ${reading.sources.join('+') || 'none'})`);
@@ -426,6 +542,10 @@ if (strays.length) {
426
542
  console.error(' ^ a COLLISION, not a wrong turn: neither reading carries a chrome');
427
543
  console.error(' label, so both are structure with no name and the fingerprint has');
428
544
  console.error(' nothing left to tell two list screens apart.');
545
+ } else if (underRead) {
546
+ console.error(' ^ UNDER-READ, not a wrong turn: this reading stayed below the token');
547
+ console.error(' floor after every retry, so it never saw enough of its screen to');
548
+ console.error(' resemble one. Two screens read this thinly are the same screen.');
429
549
  }
430
550
  if (named.length) console.error(` names: ${named.map((t) => t.slice(t.indexOf('"'), t.lastIndexOf('"') + 1)).join(' ')}`);
431
551
  }
@@ -433,6 +553,10 @@ if (strays.length) {
433
553
  console.error(`\n${collisions} of ${strays.length} are fingerprint collisions. That is this harness's own subject,`);
434
554
  console.error('not a tour fault: a reading whose chrome label went missing cannot establish');
435
555
  console.error('identity, and comparing it as though it could is what produced the verdict above.');
556
+ } else if (strays.every((x) => sparseAt.has(`${x.reading.name}|${x.reading.round}`))) {
557
+ console.error('\nEvery stray above was UNDER-READ, so this says nothing about the tour or the');
558
+ console.error('fingerprint — the harness scored a look that was too short. The retries are in');
559
+ console.error('SPARSE_READ_RETRIES; a runner that needs more than they allow is the finding.');
436
560
  } else {
437
561
  console.error('\nThat is the tour going somewhere unintended, not the fingerprint drifting, and');
438
562
  console.error('measuring it as either distribution poisons both ends. Fix the tour — a tap that');
@@ -13,6 +13,46 @@ import * as plist from './plist.js';
13
13
 
14
14
  const run = promisify(execFile);
15
15
 
16
+ /**
17
+ * How long a `simctl` verb may take before we stop waiting.
18
+ *
19
+ * **It was 20s, and that was below what a loaded runner actually needs.** Item
20
+ * 142 recorded `simctl launch` "taking 47-55s" on a hosted runner and filed it
21
+ * under a boot that had not finished; the launches were real and the budget was
22
+ * simply shorter than they were. The same 20s sat on `openurl`, which is the
23
+ * failure class that item listed four runs of, and on `terminate`. One number,
24
+ * three symptoms, and every one of them read as the command refusing rather
25
+ * than as us leaving.
26
+ *
27
+ * 90s is chosen against that measurement — comfortably past the observed 55s
28
+ * worst case — and not for feel. A `simctl` verb that has not returned in
29
+ * ninety seconds is genuinely wrong, and says so below instead of being
30
+ * indistinguishable from a rejection.
31
+ */
32
+ const SIMCTL_TIMEOUT_MS = 90_000;
33
+
34
+ /** A local file decode, which owes nothing to device latency. */
35
+ const PLUTIL_TIMEOUT_MS = 20_000;
36
+
37
+ /**
38
+ * What actually went wrong, including the case that has been invisible.
39
+ *
40
+ * On a timeout `execFile` kills the child, so `stderr` is EMPTY and `message`
41
+ * is the bare "Command failed: xcrun simctl ..." — which reads exactly like
42
+ * simctl rejecting the request. Three separate investigations have started from
43
+ * that sentence and gone looking for a broken device. The timeout has to name
44
+ * itself, or the next one starts in the same wrong place.
45
+ */
46
+ function simctlFailure(err, what) {
47
+ const detail = (err.stderr || '').trim().split('\n').filter(Boolean).pop();
48
+ if (detail) return `${what}: ${detail}`;
49
+ if (err.killed || err.signal === 'SIGTERM') {
50
+ return `${what}: simctl did not return within ${Math.round(SIMCTL_TIMEOUT_MS / 1000)}s`
51
+ + ' (killed by simframe, not refused by simctl — the host is loaded or the device is not answering)';
52
+ }
53
+ return `${what}: ${err.message}`;
54
+ }
55
+
16
56
  // `simctl list` costs ~130ms, which would otherwise dominate every warm read,
17
57
  // so the parsed list is cached for a few seconds.
18
58
  const DEVICE_CACHE_MS = 4000;
@@ -240,24 +280,33 @@ async function launchApp(udid, bundleId, { args = [], env = {}, terminateFirst =
240
280
  for (const [k, v] of Object.entries(env)) childEnv[`SIMCTL_CHILD_${k}`] = String(v);
241
281
  try {
242
282
  await run('xcrun', ['simctl', 'launch', udid, bundleId, ...args.map(String)], {
243
- timeout: 20_000,
283
+ timeout: SIMCTL_TIMEOUT_MS,
244
284
  env: childEnv,
245
285
  });
246
286
  } catch (err) {
247
287
  // execFile's message is just "Command failed: ..." with simctl's actual
248
288
  // complaint left in stderr. A CI run failed here and said nothing about
249
289
  // why, which is the same sin as a silent fallback.
250
- const detail = (err.stderr || '').trim().split('\n').filter(Boolean).pop();
251
- throw new Error(detail ? `could not launch ${bundleId}: ${detail}` : `could not launch ${bundleId}: ${err.message}`);
290
+ throw new Error(simctlFailure(err, `could not launch ${bundleId}`));
252
291
  }
253
292
  }
254
293
 
255
294
  async function terminateApp(udid, bundleId) {
256
- await run('xcrun', ['simctl', 'terminate', udid, bundleId], { timeout: 20_000 });
295
+ try {
296
+ await run('xcrun', ['simctl', 'terminate', udid, bundleId], { timeout: SIMCTL_TIMEOUT_MS });
297
+ } catch (err) {
298
+ throw new Error(simctlFailure(err, `could not terminate ${bundleId}`));
299
+ }
257
300
  }
258
301
 
259
302
  async function openUrl(udid, url) {
260
- await run('xcrun', ['simctl', 'openurl', udid, url], { timeout: 20_000 });
303
+ // The same budget and the same reporting as `launch`, because it was the same
304
+ // 20s and it is the failure class item 142 counted four runs of.
305
+ try {
306
+ await run('xcrun', ['simctl', 'openurl', udid, url], { timeout: SIMCTL_TIMEOUT_MS });
307
+ } catch (err) {
308
+ throw new Error(simctlFailure(err, 'could not open the url'));
309
+ }
261
310
  }
262
311
 
263
312
  /**
@@ -305,10 +354,9 @@ async function setPermission(udid, action, service, bundleId) {
305
354
  const args = ['simctl', 'privacy', udid, verb, service];
306
355
  if (bundleId) args.push(bundleId);
307
356
  try {
308
- await run('xcrun', args, { timeout: 20_000 });
357
+ await run('xcrun', args, { timeout: SIMCTL_TIMEOUT_MS });
309
358
  } catch (err) {
310
- const detail = (err.stderr || '').trim().split('\n').filter(Boolean).pop();
311
- throw new Error(`could not ${verb} ${service}: ${detail || err.message}`);
359
+ throw new Error(simctlFailure(err, `could not ${verb} ${service}`));
312
360
  }
313
361
  return `${verb === 'reset' ? 'reset' : verb + 'ed'} ${service}${bundleId ? ` for ${bundleId}` : ''}`;
314
362
  }
@@ -422,8 +470,12 @@ async function appContainer(udid, bundleId) {
422
470
  * the note at the top of plist.js.
423
471
  */
424
472
  async function readPropertyList(file) {
473
+ // Its own budget, deliberately not the simctl one. This reads a local file
474
+ // and never speaks to a device, so it has none of the latency the simctl
475
+ // budget exists to absorb — and a plist that takes twenty seconds to decode
476
+ // is a problem worth hearing about promptly.
425
477
  const { stdout } = await run('plutil', ['-convert', 'xml1', '-o', '-', file], {
426
- timeout: 20_000,
478
+ timeout: PLUTIL_TIMEOUT_MS,
427
479
  maxBuffer: 64 * 1024 * 1024,
428
480
  });
429
481
  return plist.parse(stdout);