simframe 0.11.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -636,9 +636,11 @@ simframe strip --count=6 # contact sheet, for an animation
636
636
  simframe doctor --strict # any degraded layer is a non-zero exit
637
637
  simframe escalations # why simframe still needs a model, by reason
638
638
  simframe escalations --session # ...this agent only, not every agent on the device
639
+ simframe supervisions # local supervisor rulings, and what came of each
639
640
  simframe hpi # speed and accuracy against a human baseline
640
641
  simframe baseline record settings-larger-text --runs=5 # record the human
641
642
  simframe input reset # rebuild the HID session, without restarting anything
643
+ simframe revive # power-cycle a wedged device: stop, shutdown, boot, start, reset input
642
644
  simframe start / status / stop [--force] / devices
643
645
  simframe ui --device=emulator-5554 # or export SIMFRAME_DEVICE once
644
646
  ```
@@ -648,6 +650,48 @@ first: the space form set the flag to `true` and then resolved a device named
648
650
  "true", which is a poor answer to a flag `doctor`'s own advice tells you to
649
651
  type.
650
652
 
653
+ ### When the simulator stops rendering
654
+
655
+ A simulator driven hard for several minutes can stop rendering: capture fails,
656
+ `simctl screenshot` fails too, and the daemon's own recoveries — re-resolving the
657
+ display port, then rebinding the device — do not help. It reports the state
658
+ rather than acting on it, because a capture loop that rebooted the device it was
659
+ watching would be a tool reaching for the mains:
660
+
661
+ ```
662
+ capture is wedged and both recoveries are spent (2 port re-resolves, 2 device
663
+ rebinds, no frame since). This needs the device restarted —
664
+ `simframe revive --device=<udid>`. Backing off until a frame arrives.
665
+ ```
666
+
667
+ `simframe revive` is that restart, in the order that matters — stop the daemon,
668
+ shut the device down, boot it and *wait for the boot to finish*, start capture,
669
+ rebuild the HID session — and it ends by checking frames are flowing again
670
+ rather than by reporting that the steps ran. It is a command and not a
671
+ behaviour: the decision stays yours.
672
+
673
+ ### Reading what the local supervisor decided
674
+
675
+ With `SIMFRAME_SUPERVISOR=apple`, every consultation is written to
676
+ `~/.simframe/<udid>/supervisions.jsonl` with what the executor observed
677
+ afterwards, which is what makes a ruling scoreable rather than merely recorded.
678
+ `simframe supervisions` reads it:
679
+
680
+ ```
681
+ 14 supervisor rulings on iPhone 17 Pro
682
+
683
+ stop -> stopped 7
684
+ wait -> recovered 5
685
+ wait -> still_failed 2
686
+
687
+ sourced: model 14
688
+ median latency: 1440ms
689
+ edges the graph had timed: 0/14 — 14 ruling(s) are on edges with no p95
690
+ ```
691
+
692
+ That last line is the honest one: a step that failed is usually a step that has
693
+ never succeeded on that edge, so the graph has no timing to compare against.
694
+
651
695
  ### Keeping a session cheap
652
696
 
653
697
  The expensive part of driving a simulator with an agent is not the tapping, it
@@ -167,6 +167,9 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
167
167
  /// session survives. Input is warmed too, because a session that outlived
168
168
  /// its device is dead anyway.
169
169
  public func reattachDevice(udid: String?) throws -> DeviceInfo {
170
+ // Before the port reference goes, not after: unregistering needs the
171
+ // port it was registered on.
172
+ unregisterChangeCallback()
170
173
  device = nil
171
174
  display = nil
172
175
  return try attach(udid: udid)
@@ -183,6 +186,10 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
183
186
  }
184
187
 
185
188
  private func resolveDisplay(on device: NSObject, warmInput: Bool) throws -> DeviceInfo {
189
+ // This re-walks the same `ioPorts` array and frequently finds the *same*
190
+ // port object, so replacing `display` without releasing the callback on
191
+ // the outgoing one is how registrations accumulated on a single port.
192
+ unregisterChangeCallback()
186
193
  guard let io = device.value(forKey: "io") as? NSObject,
187
194
  let ports = io.value(forKey: "ioPorts") as? [NSObject] else {
188
195
  throw PrivateAPIError.noDisplayPort
@@ -359,6 +366,17 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
359
366
  /// to dismiss as broken.
360
367
  public func observeChanges(_ handler: @escaping () -> Void) throws {
361
368
  guard let display else { throw PrivateAPIError.noDisplayPort }
369
+ // Registration is not idempotent and this used to be called as though it
370
+ // were. Every call minted a fresh UUID and left the previous
371
+ // registration live, and the recovery loop calls it on every attempt:
372
+ // one log from 2026-09-12 shows **670 port re-resolves**, which is up to
373
+ // 670 damage callbacks registered on one port, each invoked per redraw
374
+ // (~52/s while an app switches).
375
+ //
376
+ // That is the best available explanation for a wedge that gets worse the
377
+ // longer it runs and that only a device restart cures — and it means the
378
+ // recovery was a cause as well as a response. Unregister first, always.
379
+ unregisterChangeCallback()
362
380
  let sel = NSSelectorFromString("registerCallbackWithUUID:damageRectanglesCallback:")
363
381
  guard display.responds(to: sel), let imp = display.method(for: sel) else {
364
382
  throw PrivateAPIError.frameworksUnavailable("damageRectanglesCallback missing")
@@ -371,7 +389,13 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
371
389
  unsafeBitCast(imp, to: RegFn.self)(display, sel, uuid, block as AnyObject)
372
390
  }
373
391
 
374
- public func detach() {
392
+ /// Drop the damage callback we registered, if we registered one.
393
+ ///
394
+ /// Safe to call when there is nothing to drop, which is what lets both
395
+ /// `observeChanges` and the port-replacement paths call it unconditionally.
396
+ /// Unregisters against the port it was registered on — the one currently in
397
+ /// `display` — so it must run *before* that reference is replaced.
398
+ func unregisterChangeCallback() {
375
399
  if let display, let uuid = changeUUID {
376
400
  let sel = NSSelectorFromString("unregisterDamageRectanglesCallbackWithUUID:")
377
401
  if display.responds(to: sel), let imp = display.method(for: sel) {
@@ -381,6 +405,10 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
381
405
  }
382
406
  changeCallback = nil
383
407
  changeUUID = nil
408
+ }
409
+
410
+ public func detach() {
411
+ unregisterChangeCallback()
384
412
  display = nil
385
413
  attached = nil
386
414
  }
@@ -75,6 +75,25 @@ public struct CaptureRecovery {
75
75
  reattaches >= Self.stalledAfterReattaches && rebinds < Self.maxRebinds
76
76
  }
77
77
 
78
+ /// Both rungs of the ladder have been tried and no frame has arrived since.
79
+ ///
80
+ /// What happens after the ladder runs out was left implicit, and the
81
+ /// implicit answer was "go back to the bottom rung forever". Measured from a
82
+ /// real wedge: **670 re-resolves against 42 rebinds** in one log. Once
83
+ /// `rebinds` hits `maxRebinds` — and it only resets on a real frame —
84
+ /// `needsRebind` is false for good, so every sixth failure re-resolved a
85
+ /// port that two rebinds had already proved was not the problem, at 600ms a
86
+ /// go, each time logging a message that calls the condition "usually
87
+ /// transient".
88
+ ///
89
+ /// That is why a wedged device reads as the tool hanging rather than the
90
+ /// tool reporting. The state is unchanged in spirit — the daemon still only
91
+ /// reports, and restarting the device stays the operator's call — but it can
92
+ /// stop pretending it has something left to try.
93
+ public var recoveryExhausted: Bool {
94
+ rebinds >= Self.maxRebinds && reattaches >= Self.stalledAfterReattaches
95
+ }
96
+
78
97
  public mutating func captureSucceeded() {
79
98
  consecutiveFailures = 0
80
99
  // A real frame is the only evidence that health is back. Resetting this
@@ -192,6 +192,8 @@ case "run":
192
192
  var dirty = true
193
193
  var recovery = CaptureRecovery()
194
194
  var stalledSince: Double?
195
+ /// Said once per stall episode, not once per failed read.
196
+ var announcedExhausted = false
195
197
  var lastCapture = 0.0
196
198
  var frames = 0
197
199
  var lastReport = Date().timeIntervalSince1970
@@ -495,6 +497,7 @@ case "run":
495
497
  // back on its own, which nobody would otherwise know.
496
498
  try? store.writeCaptureHealth(nil)
497
499
  stalledSince = nil
500
+ announcedExhausted = false
498
501
  FileHandle.standardError.write(
499
502
  "simframed: capture recovered on its own\n".data(using: .utf8)!)
500
503
  }
@@ -510,7 +513,19 @@ case "run":
510
513
  // by restarting the daemon. Reporting a failure loudly is
511
514
  // right; never recovering from it is not, so re-resolve the
512
515
  // port and re-arm the damage callback.
513
- if due {
516
+ // Nothing left to try is a thing to say once, not a
517
+ // rung to keep pulling. See `recoveryExhausted`.
518
+ if due && recovery.recoveryExhausted {
519
+ if !announcedExhausted {
520
+ announcedExhausted = true
521
+ FileHandle.standardError.write(
522
+ ("simframed: capture is wedged and both recoveries are spent"
523
+ + " (\(recovery.reattaches) port re-resolves, \(recovery.rebinds) device rebinds,"
524
+ + " no frame since). This needs the device restarted —"
525
+ + " `simframe revive --device=<udid>`. Backing off until a frame arrives.\n")
526
+ .data(using: .utf8)!)
527
+ }
528
+ } else if due {
514
529
  let onDamage = { lock.lock(); dirty = true; lock.unlock() }
515
530
  // Escalate rather than repeat. Two successful re-resolves
516
531
  // with no frame between them means the port was never the
@@ -552,7 +567,10 @@ case "run":
552
567
  "reason": "\(error)",
553
568
  ])
554
569
  }
555
- Thread.sleep(forTimeInterval: 0.5)
570
+ // Back off once there is nothing left to attempt. Half a
571
+ // second forever on a dead display is pure heat, and the
572
+ // log it produced buried everything else in the file.
573
+ Thread.sleep(forTimeInterval: recovery.recoveryExhausted ? 5.0 : 0.5)
556
574
  }
557
575
  }
558
576
  }
@@ -397,4 +397,45 @@ final class CaptureRecoveryTests: XCTestCase {
397
397
  XCTAssertEqual(recovery.consecutiveFailures, 2, "the run is not cleared by an attempt that did not work")
398
398
  XCTAssertTrue(recovery.captureFailed(), "so the next failure tries again immediately")
399
399
  }
400
+
401
+ /// What happens when the ladder runs out, which was previously implicit —
402
+ /// and the implicit answer was "start again at the bottom, forever".
403
+ func testAnExhaustedLadderStopsPretendingItHasSomethingLeft() {
404
+ let platform = StubPlatform()
405
+ var recovery = CaptureRecovery(threshold: 1)
406
+ XCTAssertFalse(recovery.recoveryExhausted, "nothing has been tried yet")
407
+
408
+ // Two re-resolves that each succeed and change nothing: this is the
409
+ // pathology, not a hypothetical. The port hands back a fresh descriptor
410
+ // and every read still fails.
411
+ for _ in 0..<CaptureRecovery.stalledAfterReattaches {
412
+ _ = recovery.captureFailed()
413
+ guard case .success = recovery.reattach(platform: platform, onDamage: {}) else {
414
+ return XCTFail("the stub re-resolves")
415
+ }
416
+ }
417
+ XCTAssertTrue(recovery.needsRebind, "so the escalation is due")
418
+ XCTAssertFalse(recovery.recoveryExhausted, "but the ladder has a second rung")
419
+
420
+ for _ in 0..<CaptureRecovery.maxRebinds {
421
+ _ = recovery.captureFailed()
422
+ guard case .success = recovery.rebind(platform: platform, udid: "UDID", onDamage: {}) else {
423
+ return XCTFail("the stub rebinds")
424
+ }
425
+ }
426
+ XCTAssertFalse(recovery.needsRebind, "the rebinds are spent")
427
+ XCTAssertTrue(recovery.recoveryExhausted, "and so is the ladder")
428
+
429
+ // Measured from a real wedge: 670 port re-resolves against 42 rebinds in
430
+ // one log, because this state fell back to the bottom rung every sixth
431
+ // failure at 600ms a go, each time logging "usually transient".
432
+ _ = recovery.captureFailed()
433
+ XCTAssertTrue(recovery.recoveryExhausted, "more failures do not restore an option")
434
+
435
+ // A real frame is the only thing that resets it, exactly as for the rest
436
+ // of this machine — a re-resolve must never look like recovery.
437
+ recovery.captureSucceeded()
438
+ XCTAssertFalse(recovery.recoveryExhausted, "a frame is the only evidence health is back")
439
+ XCTAssertFalse(recovery.isStalled)
440
+ }
400
441
  }
@@ -121,11 +121,19 @@ func serve() async {
121
121
  are not being asked what to do, only whether this can proceed. Answer with \
122
122
  the decision alone.
123
123
  """
124
- // A throwaway session up front so the first real judgement does not pay
125
- // model load — measured at ~880ms against ~600ms warm.
126
- _ = try? await LanguageModelSession(instructions: instructions)
127
- .respond(to: "Step: warm\nIt failed with: warm", generating: Judgement.self)
128
- emit(["ready": true])
124
+ // Warm up front so the first real judgement does not pay model load —
125
+ // measured at ~880ms against ~600ms warm.
126
+ //
127
+ // `prewarm()` is the supported way and this used to be a throwaway
128
+ // `respond`, which loaded the same weights the hard way and paid a whole
129
+ // generation to do it. Warming a session we then discard works because
130
+ // residency is process- and system-managed: the expensive part outlives the
131
+ // session, which is also why fresh-session-per-judgement is affordable.
132
+ LanguageModelSession(instructions: instructions).prewarm()
133
+ // The window, reported rather than assumed. 4,096 has been a documented
134
+ // constant we repeated; since 26.4 it is queryable, so it is now read from
135
+ // the model and handed to the caller, who prints it in `doctor`.
136
+ emit(["ready": true, "contextSize": SystemLanguageModel.default.contextSize])
129
137
  while let line = readLine(strippingNewline: true) {
130
138
  if line.isEmpty { continue }
131
139
  guard let data = line.data(using: .utf8),
@@ -165,8 +173,35 @@ func serve() async {
165
173
  "decision": out.content.decision.rawValue,
166
174
  "ms": Int(Date().timeIntervalSince(started) * 1000),
167
175
  ])
176
+ } catch let err as LanguageModelSession.GenerationError {
177
+ // Named, not merely stringified, because the cases mean different
178
+ // things to whoever reads the log — and because a string match on
179
+ // an error message is a classification that silently becomes
180
+ // "unknown" the day Apple rewords it.
181
+ //
182
+ // Deliberately NOT a retry policy. The research that asked for this
183
+ // warned that a blanket retry burns battery on the permanent cases,
184
+ // and checking found we never retry at all: `ask` resolves null on
185
+ // any error and `judge` refuses anything outside the three words. So
186
+ // what these buy is a diagnosis, which is what was actually missing
187
+ // — every failure reached the log as "the supervisor did not answer".
188
+ let kind: String
189
+ switch err {
190
+ // Our own bug if it appears: the caller clips every field and the
191
+ // worst case those caps allow measures 1,918 tokens of 4,096.
192
+ case .exceededContextWindowSize: kind = "context-window"
193
+ // Permanent for this input. Retrying cannot help and it says so.
194
+ case .guardrailViolation: kind = "guardrail"
195
+ case .unsupportedLanguageOrLocale: kind = "locale"
196
+ // A session takes one request at a time. The line protocol here
197
+ // serialises them and each gets a fresh session, so this should be
198
+ // unreachable — worth knowing loudly if it ever is not.
199
+ case .rateLimited: kind = "rate-limited"
200
+ default: kind = "generation"
201
+ }
202
+ emit(["error": "\(err)", "kind": kind])
168
203
  } catch {
169
- emit(["error": "\(error)"])
204
+ emit(["error": "\(error)", "kind": "unknown"])
170
205
  }
171
206
  }
172
207
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "simframe",
3
- "version": "0.11.0",
3
+ "version": "0.12.0",
4
4
  "mcpName": "io.github.lvlrSajjad/simframe",
5
5
  "description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
6
6
  "keywords": [
@@ -53,6 +53,15 @@ export const ALLOWED = [
53
53
  /^com\.google\./i,
54
54
  /^org\.swift\./i,
55
55
  /^org\.json\./i,
56
+ // Build-tool namespaces, not app identifiers — and they arrive by the
57
+ // hundred the moment an Android project exists. `org.jetbrains.kotlin.android`
58
+ // is a Gradle plugin id, `org.jetbrains.kotlin:kotlin-gradle-plugin` a Maven
59
+ // coordinate, and `org.gradle.jvmargs` is a *property name* that merely looks
60
+ // like a bundle id. Added when the React Native testbed landed and flagged
61
+ // four of them; same judgement as the two lines above, which are Swift's and
62
+ // JSON's own namespaces rather than anybody's app.
63
+ /^org\.jetbrains\./i,
64
+ /^org\.gradle\./i,
56
65
  /^com\.facebook\./i, // idb, a reference implementation named in the docs
57
66
  /^com\.example\./i,
58
67
  /^com\.acme\./i,
@@ -0,0 +1,79 @@
1
+ #!/bin/bash
2
+ # The `integration` job, run here instead of there.
3
+ #
4
+ # One round trip on a hosted runner is 30+ minutes, and cancel-in-progress means
5
+ # a second push while you wait throws the answer away. Most of what that job
6
+ # asserts is reproducible on a developer's own simulator in a couple of minutes,
7
+ # so there is no reason to learn it from GitHub.
8
+ #
9
+ # What it cannot reproduce is the runner's *speed*: this machine settles a
10
+ # screen in a few hundred milliseconds where a loaded hosted runner has taken
11
+ # 45s for a two-step flow. So a green run here means "the code is right", not
12
+ # "CI will pass" — see DEFERRED 95. Everything else that has gone red, including
13
+ # the fingerprint collision that took an evening to find, would have shown up in
14
+ # this script.
15
+ #
16
+ # ./scripts/ci-integration-local.sh <udid>
17
+ #
18
+ # Assumes the device is booted and `simframe start` has been run on it, exactly
19
+ # as the job's earlier steps do.
20
+ set -u
21
+ DEVICE="${1:?usage: ci-integration-local.sh <udid>}"
22
+ cd "$(dirname "$0")/.." || exit 1
23
+ fails=0
24
+ step() { printf '\n=== %s ===\n' "$1"; }
25
+ ok() { printf 'ok %s\n' "$1"; }
26
+ bad() { printf 'FAIL %s\n' "$1"; fails=$((fails+1)); }
27
+
28
+ export SIMFRAME_STRICT=1
29
+
30
+ step "Every layer is the good one, or this fails"
31
+ node src/cli.js doctor --json --device="$DEVICE" > /tmp/doctor-local.json 2>/dev/null
32
+ node -e '
33
+ const d = JSON.parse(require("fs").readFileSync("/tmp/doctor-local.json", "utf8"));
34
+ const want = { "capture.engine": "simframed", "ocr.available": true };
35
+ let failed = false;
36
+ for (const [k, v] of Object.entries(want)) {
37
+ const good = d[k] === v;
38
+ console.log(`${good ? "ok " : "FAIL"} ${k} = ${JSON.stringify(d[k])} (want ${JSON.stringify(v)})`);
39
+ if (!good) failed = true;
40
+ }
41
+ for (const key of ["input.driver", "ax.driver"]) {
42
+ const good = d[key] === "simframed";
43
+ console.log(`${good ? "ok " : "FAIL"} ${key} = ${JSON.stringify(d[key])} (want "simframed")`);
44
+ if (!good) failed = true;
45
+ }
46
+ if (d.warnings > 0) {
47
+ console.log(`\n${d.warnings} degraded layer(s):`);
48
+ for (const c of d.checks.filter((c) => c.level !== "ok")) console.log(` ${c.level} ${c.name}: ${c.detail}`);
49
+ }
50
+ process.exit(failed ? 1 : 0);
51
+ ' && ok "doctor: every layer is the daemon" || bad "doctor"
52
+
53
+ step "A step runs, and capture notices the screen change"
54
+ read_state() {
55
+ node src/cli.js state --device="$DEVICE" --json \
56
+ | node -e 'const s=JSON.parse(require("fs").readFileSync(0,"utf8")); process.stdout.write(s.hash+" "+s.seq)'
57
+ }
58
+ printf '[{"button":"home"},{"settle":true}]\n' > /tmp/reset-local.json
59
+ node src/cli.js do /tmp/reset-local.json --device="$DEVICE" >/dev/null 2>&1
60
+ BEFORE=$(read_state)
61
+ printf '[{"openUrl":"https://example.com"},{"settle":true}]\n' > /tmp/flow-local.json
62
+ if node src/cli.js do /tmp/flow-local.json --device="$DEVICE" >/dev/null 2>&1; then
63
+ AFTER=$(read_state)
64
+ if [ "${BEFORE%% *}" != "${AFTER%% *}" ]; then ok "frame hash changed: ${BEFORE%% *} -> ${AFTER%% *}"
65
+ else bad "a step ran cleanly but capture saw no change"; fi
66
+ else
67
+ bad "openUrl flow failed"
68
+ fi
69
+
70
+ step "The memory layer — screen map, refs, graph, verdicts, flows"
71
+ node scripts/ci-memory.mjs --device="$DEVICE" && ok "ci-memory" || bad "ci-memory"
72
+
73
+ step "The fingerprint distributions, measured and bounded"
74
+ node scripts/eval-fingerprint.mjs --tour=test/tours/device-native.json --rounds=3 \
75
+ --device="$DEVICE" --out=/tmp/fp-local-ci.json --label=local-ci >/tmp/fp-out.txt 2>&1 \
76
+ && ok "fingerprint distributions" || { bad "fingerprint distributions"; tail -25 /tmp/fp-out.txt; }
77
+
78
+ printf '\n=== %s failure(s) ===\n' "$fails"
79
+ exit $((fails > 0))