simframe 0.6.0 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -130,6 +130,17 @@ public final class AccessibilityBridge {
130
130
  ? device.value(forKey: "accessibilityPlatformTranslationToken")
131
131
  : nil
132
132
 
133
+ // Guarded like its neighbours below, and for a sharper reason than
134
+ // tidiness: `setValue(_:forKey:)` on a key the class does not have
135
+ // raises `NSUnknownKeyException`, which is an Objective-C exception and
136
+ // therefore uncatchable from Swift. An Xcode that renames this property
137
+ // would not degrade the accessibility layer, it would crash the daemon
138
+ // — capture, input and OCR with it — which is the degrade-rather-than-
139
+ // fail rule exactly inverted, in the one file most likely to be
140
+ // invalidated by an upgrade.
141
+ guard shared.responds(to: NSSelectorFromString("setBridgeTokenDelegate:")) else {
142
+ throw AccessibilityError.unavailable("AXPTranslator has no bridgeTokenDelegate to install onto")
143
+ }
133
144
  shared.setValue(delegate, forKey: "bridgeTokenDelegate")
134
145
  if shared.responds(to: NSSelectorFromString("setSupportsDelegateTokens:")) {
135
146
  shared.setValue(true, forKey: "supportsDelegateTokens")
@@ -156,7 +167,16 @@ public final class AccessibilityBridge {
156
167
  .takeUnretainedValue() as? NSObject else {
157
168
  throw AccessibilityError.unavailable("the frontmost application did not translate to an element")
158
169
  }
159
- delegate.resetTimeouts()
170
+ // A ticket per read rather than one counter reset per read.
171
+ //
172
+ // The caller bounds how long it will *wait*, not how long this runs, so
173
+ // an abandoned read can still be walking when the next one starts. With
174
+ // a single shared counter, one call's reset cleared timeouts the other
175
+ // had already accumulated — and a tree that had lost subtrees came back
176
+ // claiming to be whole, which is precisely the failure this reporting
177
+ // exists to prevent.
178
+ let ticket = delegate.beginRead()
179
+ defer { delegate.endRead(ticket) }
160
180
  var out: [AXNode] = []
161
181
  var cut: String?
162
182
  let deadline = Date().addingTimeInterval(budget)
@@ -164,7 +184,7 @@ public final class AccessibilityBridge {
164
184
  // A guest that missed the deadline answers `emptyResponse`, which makes
165
185
  // the subtree below it look genuinely childless. Nothing in the nodes
166
186
  // can show that, so the count has to.
167
- let missed = delegate.timeouts
187
+ let missed = delegate.timeouts(for: ticket)
168
188
  if cut == nil, missed > 0 {
169
189
  cut = "\(missed) request(s) to the device timed out, so part of the tree is missing"
170
190
  }
@@ -192,6 +212,7 @@ public final class AccessibilityBridge {
192
212
  /// The pid of the app currently frontmost, or nil when the bridge cannot say.
193
213
  public func frontmostPid() -> Int32? {
194
214
  guard let app = frontmostApplication() else { return nil }
215
+ guard app.responds(to: NSSelectorFromString("pid")) else { return nil }
195
216
  return (app.value(forKey: "pid") as? NSNumber)?.int32Value
196
217
  }
197
218
 
@@ -318,27 +339,40 @@ private final class BridgeDelegate: NSObject {
318
339
  private let timeout: TimeInterval
319
340
  private let queue = DispatchQueue(label: "simframe.accessibility.bridge")
320
341
  private let counter = NSLock()
321
- private var timedOut = 0
342
+ /// Timeouts per in-flight read, so two overlapping reads cannot clear or
343
+ /// inherit each other's count. A timeout with no read in flight belongs to
344
+ /// an abandoned one and is dropped rather than charged to a stranger.
345
+ private var timedOut: [Int: Int] = [:]
346
+ private var nextTicket = 0
322
347
 
323
348
  init(device: NSObject, timeout: TimeInterval) {
324
349
  self.device = device
325
350
  self.timeout = timeout
326
351
  }
327
352
 
328
- /// How many requests the device failed to answer since the last reset.
329
- var timeouts: Int {
353
+ func beginRead() -> Int {
354
+ counter.lock(); defer { counter.unlock() }
355
+ nextTicket += 1
356
+ timedOut[nextTicket] = 0
357
+ return nextTicket
358
+ }
359
+
360
+ /// How many requests the device failed to answer during this read.
361
+ func timeouts(for ticket: Int) -> Int {
330
362
  counter.lock(); defer { counter.unlock() }
331
- return timedOut
363
+ return timedOut[ticket] ?? 0
332
364
  }
333
365
 
334
- func resetTimeouts() {
366
+ func endRead(_ ticket: Int) {
335
367
  counter.lock(); defer { counter.unlock() }
336
- timedOut = 0
368
+ timedOut.removeValue(forKey: ticket)
337
369
  }
338
370
 
339
371
  fileprivate func recordTimeout() {
340
372
  counter.lock(); defer { counter.unlock() }
341
- timedOut += 1
373
+ // Charged to every read currently in flight: a request that timed out
374
+ // while two reads were walking cost both of them a subtree.
375
+ for key in timedOut.keys { timedOut[key, default: 0] += 1 }
342
376
  }
343
377
 
344
378
  @objc(accessibilityTranslationDelegateBridgeCallbackWithToken:)
@@ -28,6 +28,7 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
28
28
  // machine where the translation framework is missing must still capture.
29
29
  private var accessibility: AccessibilityBridge?
30
30
  private var accessibilityFailure: String?
31
+ private let bridgeLock = NSLock()
31
32
  private var simulatorKitHandle: UnsafeMutableRawPointer?
32
33
  // Read once at attach: spawning simctl per status call cost 300ms.
33
34
  private var cachedKeyboardWarning: [String]?
@@ -193,8 +194,10 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
193
194
  // process-wide translator, so it belongs to the device it was built
194
195
  // for. Binding to a different one must not inherit it.
195
196
  if self.device !== device {
197
+ bridgeLock.lock()
196
198
  accessibility = nil
197
199
  accessibilityFailure = nil
200
+ bridgeLock.unlock()
198
201
  }
199
202
  self.device = device
200
203
  let resolved = info(for: device)
@@ -229,7 +232,18 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
229
232
  try bridge().tree()
230
233
  }
231
234
 
235
+ /// Serialised, because building a bridge installs a delegate on a
236
+ /// **process-global** translator that holds it weakly.
237
+ ///
238
+ /// Two constructions racing both install; the loser's delegate is released
239
+ /// and the survivor is left holding a translator whose delegate has
240
+ /// deallocated, which answers nil for everything and sets no failure — the
241
+ /// exact silent-nil this file's header warns about. The control socket is
242
+ /// serial, but the capture loop's rebind path clears these same two fields,
243
+ /// so there is a genuine second writer.
232
244
  private func bridge() throws -> AccessibilityBridge {
245
+ bridgeLock.lock()
246
+ defer { bridgeLock.unlock() }
233
247
  if let accessibility { return accessibility }
234
248
  // A framework that is missing stays missing; re-probing it on every
235
249
  // screen read would cost a dlopen per call to learn the same thing.
@@ -39,8 +39,14 @@ public final class StubPlatform: SimulatorPlatform {
39
39
  public private(set) var resetInputCount = 0
40
40
  public func resetInput() throws { resetInputCount += 1 }
41
41
 
42
+ /// Test hook: make re-resolving the port fail, which is the case a real
43
+ /// device only reaches when something is badly wrong and therefore the one
44
+ /// hardest to observe.
45
+ public var failReattach = false
46
+
42
47
  public func reattachDisplay() throws -> DeviceInfo {
43
48
  reattachCount += 1
49
+ if failReattach { throw PrivateAPIError.noDisplayPort }
44
50
  return DeviceInfo(udid: attachedUdid ?? "STUB-0000", name: "Stub Device", runtime: "iOS 26.0")
45
51
  }
46
52
 
@@ -0,0 +1,60 @@
1
+ import Foundation
2
+ import PrivateAPI
3
+
4
+ /// When a run of failed captures means the display port itself is gone.
5
+ ///
6
+ /// The port can be torn down and rebuilt under a running daemon, and every read
7
+ /// on the old descriptor returns nil from then on. Observed on a device awake
8
+ /// and visible the whole time: six minutes of "the display surface could not be
9
+ /// read", cured instantly by restarting the daemon.
10
+ ///
11
+ /// This lives in its own type for one reason: the recovery path only ever runs
12
+ /// in the situation nobody is watching, and inline in the capture loop it could
13
+ /// not be tested at all. A teardown cannot be induced on demand, but the
14
+ /// decision to re-resolve and the act of re-resolving both can be.
15
+ public struct CaptureRecovery {
16
+ /// Roughly three seconds of failed reads at the loop's half-second back-off.
17
+ /// Long enough not to thrash on a momentary hiccup, short enough that nobody
18
+ /// watches a dead capture loop and wonders.
19
+ public static let reattachAfterFailures = 6
20
+
21
+ public private(set) var consecutiveFailures = 0
22
+ private let threshold: Int
23
+
24
+ public init(threshold: Int = CaptureRecovery.reattachAfterFailures) {
25
+ self.threshold = threshold
26
+ }
27
+
28
+ public mutating func captureSucceeded() {
29
+ consecutiveFailures = 0
30
+ }
31
+
32
+ /// Records a failure and says whether the port is now due a re-resolve.
33
+ public mutating func captureFailed() -> Bool {
34
+ consecutiveFailures += 1
35
+ return consecutiveFailures >= threshold
36
+ }
37
+
38
+ /// Re-resolve the port and re-arm the damage callback.
39
+ ///
40
+ /// Both halves matter and only one is obvious: a fresh descriptor with no
41
+ /// callback registered on it produces a daemon that has recovered and will
42
+ /// never notice another change, which looks exactly like the failure it just
43
+ /// recovered from.
44
+ public mutating func reattach(
45
+ platform: SimulatorPlatform,
46
+ onDamage: @escaping () -> Void
47
+ ) -> Result<Int, Error> {
48
+ let failures = consecutiveFailures
49
+ do {
50
+ _ = try platform.reattachDisplay()
51
+ try platform.observeChanges(onDamage)
52
+ consecutiveFailures = 0
53
+ return .success(failures)
54
+ } catch {
55
+ // Deliberately not reset: if the port cannot be re-resolved, the
56
+ // next failure should try again rather than wait for another six.
57
+ return .failure(error)
58
+ }
59
+ }
60
+ }
@@ -9,6 +9,14 @@ func flag(_ name: String) -> String? {
9
9
  return String(a.dropFirst(name.count + 3))
10
10
  }
11
11
 
12
+ /// How long a `ui` request may spend on its two reads before what has not
13
+ /// arrived is reported as missing.
14
+ ///
15
+ /// Just inside the client's own 30 s give-up, because a read the caller has
16
+ /// already abandoned is worth nothing, and stopping any earlier than that
17
+ /// converts a slow-but-correct read into a failure. See the `ui` handler.
18
+ let readBudgetMs = 25_000
19
+
12
20
  let command = args.first(where: { !$0.hasPrefix("--") }) ?? "run"
13
21
  let platform: SimulatorPlatform = args.contains("--stub") ? StubPlatform() : CoreSimulatorPlatform()
14
22
  let longEdge = Int(flag("max-dim") ?? "") ?? 700
@@ -158,11 +166,7 @@ case "run":
158
166
 
159
167
  let lock = NSLock()
160
168
  var dirty = true
161
- var consecutiveFailures = 0
162
- /// Roughly three seconds of failed reads at the loop's half-second
163
- /// back-off. Long enough not to thrash on a momentary hiccup, short
164
- /// enough that nobody watches a dead capture loop and wonders.
165
- let reattachAfterFailures = 6
169
+ var recovery = CaptureRecovery()
166
170
  var lastCapture = 0.0
167
171
  var frames = 0
168
172
  var lastReport = Date().timeIntervalSince1970
@@ -260,19 +264,29 @@ case "run":
260
264
  }
261
265
  // Bounded, because this blocks the control socket and the
262
266
  // socket is serial: a read that runs long does not just
263
- // return late, it holds up every command behind it. A CI
264
- // runner spent 28 s inside one of these. Whatever has not
265
- // arrived by the deadline is reported as missing rather
266
- // than waited for — the layer that did answer is still
267
- // worth having.
268
- let timedOut = group.wait(timeout: .now() + .seconds(12)) == .timedOut
267
+ // return late, it holds up every command behind it.
268
+ //
269
+ // The bound is "as long as the caller is willing to wait",
270
+ // not a number picked for feel. The client gives up at 30 s
271
+ // (`control.js`), so anything past that is lost either way,
272
+ // and stopping earlier only converts a slow-but-correct
273
+ // read into a failure. It was 12 s, chosen against a runner
274
+ // that once spent 28 s inside a *tree* read — a cost the
275
+ // attribute batching then removed — and a hosted runner
276
+ // where Vision has no GPU promptly failed a screen read
277
+ // with "text recognition did not finish within 12s". OCR is
278
+ // 100–400 ms on a developer's machine and evidently much
279
+ // slower on a shared one; guessing its ceiling was the
280
+ // mistake.
281
+ let timedOut = group.wait(timeout: .now() + .milliseconds(readBudgetMs)) == .timedOut
269
282
  var (axTree, axError, axMs, ocrElements, ocrError, ocrMs) = reads.snapshot()
270
283
  if timedOut {
284
+ let secs = readBudgetMs / 1000
271
285
  if wantAx, axTree.nodes.isEmpty, axError == nil {
272
- axError = "the accessibility read did not finish within 12s"
286
+ axError = "the accessibility read did not finish within \(secs)s"
273
287
  }
274
288
  if wantOcr, ocrElements.isEmpty, ocrError == nil {
275
- ocrError = "text recognition did not finish within 12s"
289
+ ocrError = "text recognition did not finish within \(secs)s"
276
290
  }
277
291
  }
278
292
  // OCR failing is fatal to a screen read in a way a missing
@@ -419,11 +433,11 @@ case "run":
419
433
  latencies.append(Double(DispatchTime.now().uptimeNanoseconds - t0) / 1e6)
420
434
  frames += 1
421
435
  lastCapture = now
422
- consecutiveFailures = 0
436
+ recovery.captureSucceeded()
423
437
  } catch {
424
- consecutiveFailures += 1
438
+ let due = recovery.captureFailed()
425
439
  FileHandle.standardError.write(
426
- "simframed: capture failed: \(error) (\(consecutiveFailures) in a row)\n".data(using: .utf8)!)
440
+ "simframed: capture failed: \(error) (\(recovery.consecutiveFailures) in a row)\n".data(using: .utf8)!)
427
441
  // The display port can be torn down and rebuilt under a
428
442
  // running daemon, and every read on the old descriptor
429
443
  // returns nil from then on. Observed on a device that was
@@ -432,18 +446,16 @@ case "run":
432
446
  // by restarting the daemon. Reporting a failure loudly is
433
447
  // right; never recovering from it is not, so re-resolve the
434
448
  // port and re-arm the damage callback.
435
- if consecutiveFailures >= reattachAfterFailures {
436
- do {
437
- _ = try platform.reattachDisplay()
438
- try platform.observeChanges {
439
- lock.lock(); dirty = true; lock.unlock()
440
- }
449
+ if due {
450
+ switch recovery.reattach(platform: platform, onDamage: {
451
+ lock.lock(); dirty = true; lock.unlock()
452
+ }) {
453
+ case .success(let after):
441
454
  lock.lock(); dirty = true; lock.unlock()
442
455
  FileHandle.standardError.write(
443
- "simframed: re-resolved the display port after \(consecutiveFailures) failed reads\n"
456
+ "simframed: re-resolved the display port after \(after) failed reads\n"
444
457
  .data(using: .utf8)!)
445
- consecutiveFailures = 0
446
- } catch {
458
+ case .failure(let error):
447
459
  FileHandle.standardError.write(
448
460
  "simframed: could not re-resolve the display port: \(error)\n".data(using: .utf8)!)
449
461
  }
@@ -268,3 +268,54 @@ final class AccessibilityElementTests: XCTestCase {
268
268
  XCTAssertNil(element.json["identifier"], "an absent identifier is absent, not null")
269
269
  }
270
270
  }
271
+
272
+ /// The recovery path that only ever runs when nobody is watching.
273
+ ///
274
+ /// A display teardown cannot be induced on demand, which is why this went
275
+ /// untested through two releases. The decision and the act can both be driven
276
+ /// with a stub, and that covers everything except the teardown itself.
277
+ final class CaptureRecoveryTests: XCTestCase {
278
+ func testAMomentaryHiccupDoesNotReattach() {
279
+ var recovery = CaptureRecovery(threshold: 6)
280
+ for _ in 0..<5 { XCTAssertFalse(recovery.captureFailed(), "five failures is a hiccup") }
281
+ XCTAssertEqual(recovery.consecutiveFailures, 5)
282
+ recovery.captureSucceeded()
283
+ XCTAssertEqual(recovery.consecutiveFailures, 0, "one good frame clears the run")
284
+ for _ in 0..<5 { XCTAssertFalse(recovery.captureFailed()) }
285
+ }
286
+
287
+ func testASustainedRunReattachesAndRearmsTheCallback() {
288
+ let platform = StubPlatform()
289
+ _ = try? platform.attach(udid: "STUB-1")
290
+ var recovery = CaptureRecovery(threshold: 3)
291
+ XCTAssertFalse(recovery.captureFailed())
292
+ XCTAssertFalse(recovery.captureFailed())
293
+ XCTAssertTrue(recovery.captureFailed(), "the third failure is due a re-resolve")
294
+
295
+ var damaged = false
296
+ let outcome = recovery.reattach(platform: platform) { damaged = true }
297
+ guard case .success(let after) = outcome else { return XCTFail("reattach should succeed on a live stub") }
298
+ XCTAssertEqual(after, 3, "it reports how many reads it lost")
299
+ XCTAssertEqual(platform.reattachCount, 1)
300
+ XCTAssertEqual(recovery.consecutiveFailures, 0)
301
+
302
+ // The callback half is the one that is easy to forget: a fresh
303
+ // descriptor with nothing registered on it gives a daemon that has
304
+ // recovered and will never notice another change.
305
+ platform.simulateChange()
306
+ XCTAssertTrue(damaged, "the damage callback was re-armed on the new descriptor")
307
+ }
308
+
309
+ func testAFailedReattachStaysDueRatherThanWaitingForAnotherSix() {
310
+ let platform = StubPlatform()
311
+ platform.failReattach = true
312
+ var recovery = CaptureRecovery(threshold: 2)
313
+ _ = recovery.captureFailed()
314
+ XCTAssertTrue(recovery.captureFailed())
315
+ guard case .failure = recovery.reattach(platform: platform, onDamage: {}) else {
316
+ return XCTFail("a stub told to fail should fail")
317
+ }
318
+ XCTAssertEqual(recovery.consecutiveFailures, 2, "the run is not cleared by an attempt that did not work")
319
+ XCTAssertTrue(recovery.captureFailed(), "so the next failure tries again immediately")
320
+ }
321
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "simframe",
3
- "version": "0.6.0",
3
+ "version": "0.6.1",
4
4
  "mcpName": "io.github.lvlrSajjad/simframe",
5
5
  "description": "Always-warm iOS Simulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
6
6
  "keywords": [
@@ -15,6 +15,7 @@
15
15
  // Swift source or a new module here and it becomes required automatically.
16
16
  import { execFileSync } from 'node:child_process';
17
17
  import fs from 'node:fs';
18
+ import os from 'node:os';
18
19
  import path from 'node:path';
19
20
  import { fileURLToPath } from 'node:url';
20
21
 
@@ -83,8 +84,34 @@ function requiredFiles() {
83
84
  }
84
85
 
85
86
  const required = requiredFiles();
86
- const packed = JSON.parse(execFileSync('npm', ['pack', '--dry-run', '--json'], { cwd: ROOT, encoding: 'utf8' }));
87
- const shipped = new Set(packed[0].files.map((f) => f.path));
87
+ // Read the tarball, not npm's description of it.
88
+ //
89
+ // This used to parse `npm pack --dry-run --json` as `packed[0].files`, which is
90
+ // npm 10's shape. npm 12 reports something else, and the check died with
91
+ // "Cannot read properties of undefined (reading 'files')" — a check that exists
92
+ // to catch a silently broken package, itself silently broken by the tool it
93
+ // asks. It had only ever run under the npm bundled with Node 18–22; the release
94
+ // job runs a newer one.
95
+ //
96
+ // `tar -tzf` on the artifact npm actually produces has no shape to change, and
97
+ // it answers the stronger question: what is *in* the file people download.
98
+ const out = fs.mkdtempSync(path.join(os.tmpdir(), 'simframe-pack-'));
99
+ const name = execFileSync('npm', ['pack', '--silent', '--pack-destination', out], { cwd: ROOT, encoding: 'utf8' })
100
+ .trim()
101
+ .split('\n')
102
+ .pop();
103
+ const listing = execFileSync('tar', ['-tzf', path.join(out, name)], { encoding: 'utf8' });
104
+ fs.rmSync(out, { recursive: true, force: true });
105
+ const shipped = new Set(
106
+ listing
107
+ .split('\n')
108
+ .map((line) => line.trim())
109
+ .filter(Boolean)
110
+ // Every path in an npm tarball is prefixed `package/`, and directories
111
+ // arrive with a trailing slash.
112
+ .filter((line) => line.startsWith('package/') && !line.endsWith('/'))
113
+ .map((line) => line.slice('package/'.length)),
114
+ );
88
115
  const missing = required.filter((f) => !shipped.has(f));
89
116
 
90
117
  console.log(`${required.length} build inputs required, ${shipped.size} files in the tarball`);
@@ -157,6 +157,19 @@ export function tokens(targets, screen) {
157
157
  }
158
158
 
159
159
  export function hashTokens(list) {
160
+ // No tokens is not an identity. It is the absence of one.
161
+ //
162
+ // sha256 of the empty string is a constant, so every unreadable screen used
163
+ // to hash to `e3b0c442…` — one identity shared by a dark screen, a screen
164
+ // whose OCR failed, and a screen read mid-transition. Different screens
165
+ // collapsing onto a single hash is a wrong *merge*, which is worse than a
166
+ // missed match: a ref numbered on one screen resolved happily on another,
167
+ // and an edge learned on one predicted the other, with a verdict of `ok`.
168
+ //
169
+ // The pixel layout hash had exactly this degeneracy and got an
170
+ // `informative()` guard for it. The structural hash did not, and the guard
171
+ // could not have helped: a constant is perfectly informative-looking.
172
+ if (!list.length) return null;
160
173
  return crypto.createHash('sha256').update(list.join('\n')).digest('hex').slice(0, 32);
161
174
  }
162
175
 
@@ -174,7 +187,11 @@ export function fingerprint(targets, screen) {
174
187
  * everything.
175
188
  */
176
189
  export function similarity(a = [], b = []) {
177
- if (!a.length && !b.length) return 1;
190
+ // Two empty token sets are not a match, they are two absences of evidence.
191
+ // Returning 1 here made unreadability *self-confirming*: two consecutive
192
+ // unreadable reads "agreed", which promoted the non-identity to a confirmed
193
+ // screen and let it be written into memory and learned as an edge.
194
+ if (!a.length || !b.length) return 0;
178
195
  const setA = new Set(a);
179
196
  const setB = new Set(b);
180
197
  let shared = 0;
package/src/graph.js CHANGED
@@ -260,7 +260,20 @@ export function record(udid, { from, action, to, kind }) {
260
260
  const claimant = nearestScreen(udid, reading)?.node;
261
261
  const target = load(udid, existing.to);
262
262
  const unclaimed = !claimant || claimant.hash === target.hash;
263
- if (unclaimed && target.hash !== to_ && reading.tokens?.length) {
263
+ // "Nothing we have stored claims this reading" is not evidence that the
264
+ // target grew a second face. It is equally consistent with this action
265
+ // being state-dependent and having gone somewhere genuinely new — and
266
+ // treating the two the same welds an unrelated screen onto a learned
267
+ // edge. Reproduced: a screen sharing *zero* tokens with the target got
268
+ // merged into it, after which landing there returned `ok` ("matches the
269
+ // outcome seen 3x before") and a flow kept walking, tapping real controls
270
+ // on a screen its plan never contained.
271
+ //
272
+ // So the reading has to positively look like the target before it is
273
+ // called a face of it. Unclaimed is a necessary condition, not a
274
+ // sufficient one.
275
+ const looksLikeTarget = resembles(target, reading);
276
+ if (unclaimed && looksLikeTarget && target.hash !== to_ && reading.tokens?.length) {
264
277
  addVariant(target, reading);
265
278
  save(udid, target);
266
279
  existing.count += 1;
@@ -293,6 +306,20 @@ export function record(udid, { from, action, to, kind }) {
293
306
  }
294
307
 
295
308
  /** What this action did last time, if we have ever seen it here. */
309
+ /**
310
+ * Does this reading look like a face of this screen, rather than a different
311
+ * screen we happen not to have stored yet?
312
+ *
313
+ * Compared against every face the screen already wears, because a screen with
314
+ * two structures is exactly the case variants exist for and a new reading may
315
+ * resemble the second one rather than the first.
316
+ */
317
+ function resembles(node, reading) {
318
+ if (!node || !reading?.tokens?.length) return false;
319
+ const faces = [node.tokens ?? [], ...(node.variants ?? []).map((v) => v.tokens ?? [])];
320
+ return faces.some((face) => face.length && fingerprint.similarity(face, reading.tokens) >= SIMILARITY_THRESHOLD);
321
+ }
322
+
296
323
  export function predict(udid, from, action) {
297
324
  const found = nearestScreen(udid, from);
298
325
  if (!found) return null;
@@ -323,10 +350,27 @@ export function forget(udid) {
323
350
  export function route(udid, fromHash, toHash, { maxDepth = 8 } = {}) {
324
351
  const start = nearestScreen(udid, fromHash);
325
352
  if (!start) return null;
326
- const goal = (h) => h === toHash;
327
- if (goal(start.node.hash)) return [];
353
+ if (start.node.hash === toHash) return [];
328
354
 
329
- const byHash = new Map(allNodes(udid).map((n) => [n.hash, n]));
355
+ // Variants have to resolve here the same way they resolve everywhere else.
356
+ //
357
+ // A screen may wear more than one structure, and an edge records whichever
358
+ // one it arrived on — so an edge whose `to` is a variant hash was a dead end
359
+ // in this search while `nearestScreen` was perfectly happy to say that hash
360
+ // *is* the node. The graph then had a route it could not find, `goto`
361
+ // answered `no-route` for somewhere it had been, and a flow that should have
362
+ // replayed from memory was re-explored instead. That is at least one
363
+ // mechanical cause of the convergence flakiness in DEFERRED.
364
+ const byHash = new Map();
365
+ for (const node of allNodes(udid)) {
366
+ byHash.set(node.hash, node);
367
+ for (const variant of node.variants ?? []) if (!byHash.has(variant.hash)) byHash.set(variant.hash, node);
368
+ }
369
+ // Canonical identity, so a variant and its screen are one place in the search
370
+ // rather than two, and reaching either counts as reaching the goal.
371
+ const canonical = (h) => byHash.get(h)?.hash ?? h;
372
+ const goalHash = canonical(toHash);
373
+ const reached = (h) => canonical(h) === goalHash;
330
374
  const seen = new Set([start.node.hash]);
331
375
  const queue = [{ hash: start.node.hash, path: [] }];
332
376
  while (queue.length) {
@@ -334,11 +378,12 @@ export function route(udid, fromHash, toHash, { maxDepth = 8 } = {}) {
334
378
  if (taken.length >= maxDepth) continue;
335
379
  const node = byHash.get(hash);
336
380
  for (const edge of node?.edges ?? []) {
337
- if (seen.has(edge.to)) continue;
381
+ const to = canonical(edge.to);
382
+ if (seen.has(to)) continue;
338
383
  const next = [...taken, edge];
339
- if (goal(edge.to)) return next;
340
- seen.add(edge.to);
341
- queue.push({ hash: edge.to, path: next });
384
+ if (reached(edge.to)) return next;
385
+ seen.add(to);
386
+ queue.push({ hash: to, path: next });
342
387
  }
343
388
  }
344
389
  return null;
@@ -375,6 +420,18 @@ function sameScreen(udid, a, b) {
375
420
  return Boolean(nodeA && nodeB && nodeA.hash === nodeB.hash);
376
421
  }
377
422
 
423
+ /**
424
+ * How many times an edge must have been observed before a mismatch counts as a
425
+ * wrong turn rather than as "we do not know yet".
426
+ *
427
+ * Two, because the difference between one and two observations is the
428
+ * difference between a coincidence and a pattern, and the cost of being wrong
429
+ * is asymmetric: halting a correct run is visible and annoying, while
430
+ * continuing one extra step past a genuinely wrong turn is caught by the next
431
+ * step's own verdict.
432
+ */
433
+ export const CONFIDENT_OBSERVATIONS = 2;
434
+
378
435
  export function verdict({ udid, prediction, before, after, kind }) {
379
436
  if (!before || !after) return { verdict: 'unverified', detail: 'no state to compare' };
380
437
  const moved = before !== after;
@@ -387,6 +444,46 @@ export function verdict({ udid, prediction, before, after, kind }) {
387
444
  return { verdict: 'no-visible-change', detail: `expected to reach a different screen (seen ${prediction.count}x)` };
388
445
  }
389
446
  if (!sameScreen(udid, prediction.to, after)) {
447
+ // One observation is not a prediction, and halting on it is what made a new
448
+ // user's *second* run worse than their first.
449
+ //
450
+ // The first run learns every edge at count 1 and cannot contradict itself,
451
+ // so it reports `unverified` throughout and completes. The second run then
452
+ // has an expectation for every step, and any screen whose identity wobbles
453
+ // — a read taken while the tree was still arriving, a screen with more than
454
+ // one legitimate structure — contradicts it and stops the run. Measured
455
+ // from the published package: run 1 halted at 3/10, runs 2 and 3 went
456
+ // 10/10. The halt was not protecting anyone from anything.
457
+ //
458
+ // So a single-observation miss is reported as what it actually is: we do
459
+ // not know yet. It must not be `unexpected-screen`, because that verdict is
460
+ // what halts a run and what "a run that reported a wrong turn never also
461
+ // reports success" is asserted over — and both of those should stay true.
462
+ // Once the same edge has been seen twice, a miss is a real wrong turn.
463
+ if (prediction.count < CONFIDENT_OBSERVATIONS) {
464
+ return {
465
+ verdict: 'unverified',
466
+ detail: `seen here once before and went somewhere else that time`
467
+ + ` — one observation is not enough to call this a wrong turn`,
468
+ weakPrediction: { count: prediction.count, to: prediction.to },
469
+ };
470
+ }
471
+ // This action has already led somewhere different at least once, so its
472
+ // destination is not a fact about the screen — it is a distribution. An app
473
+ // relaunch that lands on restored state, a list whose first row depends on
474
+ // what happened last time: these genuinely have more than one outcome, and
475
+ // calling the second one a wrong turn is calling the world wrong.
476
+ //
477
+ // The graph has always counted this as `changedOutcomes` and nothing ever
478
+ // read it.
479
+ if (prediction.changedOutcomes > 0) {
480
+ return {
481
+ verdict: 'unverified',
482
+ detail: `this action has reached ${prediction.changedOutcomes + 1} different screens from here`
483
+ + ` — its outcome is not predictable, so this is not a wrong turn`,
484
+ nondeterministic: { outcomes: prediction.changedOutcomes + 1, count: prediction.count },
485
+ };
486
+ }
390
487
  return {
391
488
  verdict: 'unexpected-screen',
392
489
  detail: `expected the screen this action reached ${prediction.count}x before, and landed somewhere else`,
package/src/screenmap.js CHANGED
@@ -13,6 +13,7 @@ import * as fingerprint from './fingerprint.js';
13
13
  import * as input from './input.js';
14
14
  import * as ocr from './ocr.js';
15
15
  import * as regions from './regions.js';
16
+ import { informative } from './refs.js';
16
17
  import * as store from './store.js';
17
18
 
18
19
  const MAP_VERSION = 6; // dates, prices and phone numbers no longer contribute labels
@@ -60,6 +61,12 @@ function loadAll(udid) {
60
61
  */
61
62
  export function recallNearest(udid, layoutHash, { tolerance = DEFAULT_TOLERANCE } = {}) {
62
63
  if (!layoutHash) return null;
64
+ // A hash of almost no set bits is a dark or uniform screen, and two of them
65
+ // are within any tolerance of each other while being evidence of nothing.
66
+ // `refs.js` documents this and guards for it; this function fed that one's
67
+ // `screenKnown` and `structuralHash` inputs without a guard of its own, so a
68
+ // near-uniform screen could hand back a different screen's element map.
69
+ if (!informative(layoutHash)) return null;
63
70
  let best = null;
64
71
  let bestDistance = Infinity;
65
72
  for (const entry of loadAll(udid)) {