simframe 0.6.0 → 0.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/native/simframed/Sources/PrivateAPI/AccessibilityBridge.swift +50 -11
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +14 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +6 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +60 -0
- package/native/simframed/Sources/simframed/main.swift +37 -25
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +51 -0
- package/package.json +1 -1
- package/scripts/check-package.mjs +29 -2
- package/scripts/ci-memory.mjs +36 -1
- package/src/fingerprint.js +48 -1
- package/src/graph.js +193 -12
- package/src/screenmap.js +8 -1
|
@@ -130,6 +130,17 @@ public final class AccessibilityBridge {
|
|
|
130
130
|
? device.value(forKey: "accessibilityPlatformTranslationToken")
|
|
131
131
|
: nil
|
|
132
132
|
|
|
133
|
+
// Guarded like its neighbours below, and for a sharper reason than
|
|
134
|
+
// tidiness: `setValue(_:forKey:)` on a key the class does not have
|
|
135
|
+
// raises `NSUnknownKeyException`, which is an Objective-C exception and
|
|
136
|
+
// therefore uncatchable from Swift. An Xcode that renames this property
|
|
137
|
+
// would not degrade the accessibility layer, it would crash the daemon
|
|
138
|
+
// — capture, input and OCR with it — which is the degrade-rather-than-
|
|
139
|
+
// fail rule exactly inverted, in the one file most likely to be
|
|
140
|
+
// invalidated by an upgrade.
|
|
141
|
+
guard shared.responds(to: NSSelectorFromString("setBridgeTokenDelegate:")) else {
|
|
142
|
+
throw AccessibilityError.unavailable("AXPTranslator has no bridgeTokenDelegate to install onto")
|
|
143
|
+
}
|
|
133
144
|
shared.setValue(delegate, forKey: "bridgeTokenDelegate")
|
|
134
145
|
if shared.responds(to: NSSelectorFromString("setSupportsDelegateTokens:")) {
|
|
135
146
|
shared.setValue(true, forKey: "supportsDelegateTokens")
|
|
@@ -156,7 +167,16 @@ public final class AccessibilityBridge {
|
|
|
156
167
|
.takeUnretainedValue() as? NSObject else {
|
|
157
168
|
throw AccessibilityError.unavailable("the frontmost application did not translate to an element")
|
|
158
169
|
}
|
|
159
|
-
|
|
170
|
+
// A ticket per read rather than one counter reset per read.
|
|
171
|
+
//
|
|
172
|
+
// The caller bounds how long it will *wait*, not how long this runs, so
|
|
173
|
+
// an abandoned read can still be walking when the next one starts. With
|
|
174
|
+
// a single shared counter, one call's reset cleared timeouts the other
|
|
175
|
+
// had already accumulated — and a tree that had lost subtrees came back
|
|
176
|
+
// claiming to be whole, which is precisely the failure this reporting
|
|
177
|
+
// exists to prevent.
|
|
178
|
+
let ticket = delegate.beginRead()
|
|
179
|
+
defer { delegate.endRead(ticket) }
|
|
160
180
|
var out: [AXNode] = []
|
|
161
181
|
var cut: String?
|
|
162
182
|
let deadline = Date().addingTimeInterval(budget)
|
|
@@ -164,7 +184,7 @@ public final class AccessibilityBridge {
|
|
|
164
184
|
// A guest that missed the deadline answers `emptyResponse`, which makes
|
|
165
185
|
// the subtree below it look genuinely childless. Nothing in the nodes
|
|
166
186
|
// can show that, so the count has to.
|
|
167
|
-
let missed = delegate.timeouts
|
|
187
|
+
let missed = delegate.timeouts(for: ticket)
|
|
168
188
|
if cut == nil, missed > 0 {
|
|
169
189
|
cut = "\(missed) request(s) to the device timed out, so part of the tree is missing"
|
|
170
190
|
}
|
|
@@ -192,6 +212,7 @@ public final class AccessibilityBridge {
|
|
|
192
212
|
/// The pid of the app currently frontmost, or nil when the bridge cannot say.
|
|
193
213
|
public func frontmostPid() -> Int32? {
|
|
194
214
|
guard let app = frontmostApplication() else { return nil }
|
|
215
|
+
guard app.responds(to: NSSelectorFromString("pid")) else { return nil }
|
|
195
216
|
return (app.value(forKey: "pid") as? NSNumber)?.int32Value
|
|
196
217
|
}
|
|
197
218
|
|
|
@@ -216,8 +237,13 @@ public final class AccessibilityBridge {
|
|
|
216
237
|
cut = cut ?? "the read ran out of time"
|
|
217
238
|
return
|
|
218
239
|
}
|
|
219
|
-
|
|
220
|
-
|
|
240
|
+
// Each node's reads hand back autoreleased objects, and a 4000-node
|
|
241
|
+
// cap means 4000 nodes' worth of them living until the whole walk
|
|
242
|
+
// returns. Draining per node keeps the peak flat.
|
|
243
|
+
autoreleasepool {
|
|
244
|
+
out.append(node(from: element, depth: depth))
|
|
245
|
+
}
|
|
246
|
+
guard let children = autoreleasepool(invoking: { attribute(element, "AXChildren") as? [NSObject] }) else { return }
|
|
221
247
|
for child in children {
|
|
222
248
|
walk(child, depth: depth + 1, into: &out, deadline: deadline, cut: &cut)
|
|
223
249
|
}
|
|
@@ -318,27 +344,40 @@ private final class BridgeDelegate: NSObject {
|
|
|
318
344
|
private let timeout: TimeInterval
|
|
319
345
|
private let queue = DispatchQueue(label: "simframe.accessibility.bridge")
|
|
320
346
|
private let counter = NSLock()
|
|
321
|
-
|
|
347
|
+
/// Timeouts per in-flight read, so two overlapping reads cannot clear or
|
|
348
|
+
/// inherit each other's count. A timeout with no read in flight belongs to
|
|
349
|
+
/// an abandoned one and is dropped rather than charged to a stranger.
|
|
350
|
+
private var timedOut: [Int: Int] = [:]
|
|
351
|
+
private var nextTicket = 0
|
|
322
352
|
|
|
323
353
|
init(device: NSObject, timeout: TimeInterval) {
|
|
324
354
|
self.device = device
|
|
325
355
|
self.timeout = timeout
|
|
326
356
|
}
|
|
327
357
|
|
|
328
|
-
|
|
329
|
-
|
|
358
|
+
func beginRead() -> Int {
|
|
359
|
+
counter.lock(); defer { counter.unlock() }
|
|
360
|
+
nextTicket += 1
|
|
361
|
+
timedOut[nextTicket] = 0
|
|
362
|
+
return nextTicket
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
/// How many requests the device failed to answer during this read.
|
|
366
|
+
func timeouts(for ticket: Int) -> Int {
|
|
330
367
|
counter.lock(); defer { counter.unlock() }
|
|
331
|
-
return timedOut
|
|
368
|
+
return timedOut[ticket] ?? 0
|
|
332
369
|
}
|
|
333
370
|
|
|
334
|
-
func
|
|
371
|
+
func endRead(_ ticket: Int) {
|
|
335
372
|
counter.lock(); defer { counter.unlock() }
|
|
336
|
-
timedOut
|
|
373
|
+
timedOut.removeValue(forKey: ticket)
|
|
337
374
|
}
|
|
338
375
|
|
|
339
376
|
fileprivate func recordTimeout() {
|
|
340
377
|
counter.lock(); defer { counter.unlock() }
|
|
341
|
-
|
|
378
|
+
// Charged to every read currently in flight: a request that timed out
|
|
379
|
+
// while two reads were walking cost both of them a subtree.
|
|
380
|
+
for key in timedOut.keys { timedOut[key, default: 0] += 1 }
|
|
342
381
|
}
|
|
343
382
|
|
|
344
383
|
@objc(accessibilityTranslationDelegateBridgeCallbackWithToken:)
|
|
@@ -28,6 +28,7 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
|
|
|
28
28
|
// machine where the translation framework is missing must still capture.
|
|
29
29
|
private var accessibility: AccessibilityBridge?
|
|
30
30
|
private var accessibilityFailure: String?
|
|
31
|
+
private let bridgeLock = NSLock()
|
|
31
32
|
private var simulatorKitHandle: UnsafeMutableRawPointer?
|
|
32
33
|
// Read once at attach: spawning simctl per status call cost 300ms.
|
|
33
34
|
private var cachedKeyboardWarning: [String]?
|
|
@@ -193,8 +194,10 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
|
|
|
193
194
|
// process-wide translator, so it belongs to the device it was built
|
|
194
195
|
// for. Binding to a different one must not inherit it.
|
|
195
196
|
if self.device !== device {
|
|
197
|
+
bridgeLock.lock()
|
|
196
198
|
accessibility = nil
|
|
197
199
|
accessibilityFailure = nil
|
|
200
|
+
bridgeLock.unlock()
|
|
198
201
|
}
|
|
199
202
|
self.device = device
|
|
200
203
|
let resolved = info(for: device)
|
|
@@ -229,7 +232,18 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
|
|
|
229
232
|
try bridge().tree()
|
|
230
233
|
}
|
|
231
234
|
|
|
235
|
+
/// Serialised, because building a bridge installs a delegate on a
|
|
236
|
+
/// **process-global** translator that holds it weakly.
|
|
237
|
+
///
|
|
238
|
+
/// Two constructions racing both install; the loser's delegate is released
|
|
239
|
+
/// and the survivor is left holding a translator whose delegate has
|
|
240
|
+
/// deallocated, which answers nil for everything and sets no failure — the
|
|
241
|
+
/// exact silent-nil this file's header warns about. The control socket is
|
|
242
|
+
/// serial, but the capture loop's rebind path clears these same two fields,
|
|
243
|
+
/// so there is a genuine second writer.
|
|
232
244
|
private func bridge() throws -> AccessibilityBridge {
|
|
245
|
+
bridgeLock.lock()
|
|
246
|
+
defer { bridgeLock.unlock() }
|
|
233
247
|
if let accessibility { return accessibility }
|
|
234
248
|
// A framework that is missing stays missing; re-probing it on every
|
|
235
249
|
// screen read would cost a dlopen per call to learn the same thing.
|
|
@@ -39,8 +39,14 @@ public final class StubPlatform: SimulatorPlatform {
|
|
|
39
39
|
public private(set) var resetInputCount = 0
|
|
40
40
|
public func resetInput() throws { resetInputCount += 1 }
|
|
41
41
|
|
|
42
|
+
/// Test hook: make re-resolving the port fail, which is the case a real
|
|
43
|
+
/// device only reaches when something is badly wrong and therefore the one
|
|
44
|
+
/// hardest to observe.
|
|
45
|
+
public var failReattach = false
|
|
46
|
+
|
|
42
47
|
public func reattachDisplay() throws -> DeviceInfo {
|
|
43
48
|
reattachCount += 1
|
|
49
|
+
if failReattach { throw PrivateAPIError.noDisplayPort }
|
|
44
50
|
return DeviceInfo(udid: attachedUdid ?? "STUB-0000", name: "Stub Device", runtime: "iOS 26.0")
|
|
45
51
|
}
|
|
46
52
|
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import Foundation
|
|
2
|
+
import PrivateAPI
|
|
3
|
+
|
|
4
|
+
/// When a run of failed captures means the display port itself is gone.
|
|
5
|
+
///
|
|
6
|
+
/// The port can be torn down and rebuilt under a running daemon, and every read
|
|
7
|
+
/// on the old descriptor returns nil from then on. Observed on a device awake
|
|
8
|
+
/// and visible the whole time: six minutes of "the display surface could not be
|
|
9
|
+
/// read", cured instantly by restarting the daemon.
|
|
10
|
+
///
|
|
11
|
+
/// This lives in its own type for one reason: the recovery path only ever runs
|
|
12
|
+
/// in the situation nobody is watching, and inline in the capture loop it could
|
|
13
|
+
/// not be tested at all. A teardown cannot be induced on demand, but the
|
|
14
|
+
/// decision to re-resolve and the act of re-resolving both can be.
|
|
15
|
+
public struct CaptureRecovery {
|
|
16
|
+
/// Roughly three seconds of failed reads at the loop's half-second back-off.
|
|
17
|
+
/// Long enough not to thrash on a momentary hiccup, short enough that nobody
|
|
18
|
+
/// watches a dead capture loop and wonders.
|
|
19
|
+
public static let reattachAfterFailures = 6
|
|
20
|
+
|
|
21
|
+
public private(set) var consecutiveFailures = 0
|
|
22
|
+
private let threshold: Int
|
|
23
|
+
|
|
24
|
+
public init(threshold: Int = CaptureRecovery.reattachAfterFailures) {
|
|
25
|
+
self.threshold = threshold
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
public mutating func captureSucceeded() {
|
|
29
|
+
consecutiveFailures = 0
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/// Records a failure and says whether the port is now due a re-resolve.
|
|
33
|
+
public mutating func captureFailed() -> Bool {
|
|
34
|
+
consecutiveFailures += 1
|
|
35
|
+
return consecutiveFailures >= threshold
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/// Re-resolve the port and re-arm the damage callback.
|
|
39
|
+
///
|
|
40
|
+
/// Both halves matter and only one is obvious: a fresh descriptor with no
|
|
41
|
+
/// callback registered on it produces a daemon that has recovered and will
|
|
42
|
+
/// never notice another change, which looks exactly like the failure it just
|
|
43
|
+
/// recovered from.
|
|
44
|
+
public mutating func reattach(
|
|
45
|
+
platform: SimulatorPlatform,
|
|
46
|
+
onDamage: @escaping () -> Void
|
|
47
|
+
) -> Result<Int, Error> {
|
|
48
|
+
let failures = consecutiveFailures
|
|
49
|
+
do {
|
|
50
|
+
_ = try platform.reattachDisplay()
|
|
51
|
+
try platform.observeChanges(onDamage)
|
|
52
|
+
consecutiveFailures = 0
|
|
53
|
+
return .success(failures)
|
|
54
|
+
} catch {
|
|
55
|
+
// Deliberately not reset: if the port cannot be re-resolved, the
|
|
56
|
+
// next failure should try again rather than wait for another six.
|
|
57
|
+
return .failure(error)
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
}
|
|
@@ -9,6 +9,14 @@ func flag(_ name: String) -> String? {
|
|
|
9
9
|
return String(a.dropFirst(name.count + 3))
|
|
10
10
|
}
|
|
11
11
|
|
|
12
|
+
/// How long a `ui` request may spend on its two reads before what has not
|
|
13
|
+
/// arrived is reported as missing.
|
|
14
|
+
///
|
|
15
|
+
/// Just inside the client's own 30 s give-up, because a read the caller has
|
|
16
|
+
/// already abandoned is worth nothing, and stopping any earlier than that
|
|
17
|
+
/// converts a slow-but-correct read into a failure. See the `ui` handler.
|
|
18
|
+
let readBudgetMs = 25_000
|
|
19
|
+
|
|
12
20
|
let command = args.first(where: { !$0.hasPrefix("--") }) ?? "run"
|
|
13
21
|
let platform: SimulatorPlatform = args.contains("--stub") ? StubPlatform() : CoreSimulatorPlatform()
|
|
14
22
|
let longEdge = Int(flag("max-dim") ?? "") ?? 700
|
|
@@ -158,11 +166,7 @@ case "run":
|
|
|
158
166
|
|
|
159
167
|
let lock = NSLock()
|
|
160
168
|
var dirty = true
|
|
161
|
-
var
|
|
162
|
-
/// Roughly three seconds of failed reads at the loop's half-second
|
|
163
|
-
/// back-off. Long enough not to thrash on a momentary hiccup, short
|
|
164
|
-
/// enough that nobody watches a dead capture loop and wonders.
|
|
165
|
-
let reattachAfterFailures = 6
|
|
169
|
+
var recovery = CaptureRecovery()
|
|
166
170
|
var lastCapture = 0.0
|
|
167
171
|
var frames = 0
|
|
168
172
|
var lastReport = Date().timeIntervalSince1970
|
|
@@ -260,19 +264,29 @@ case "run":
|
|
|
260
264
|
}
|
|
261
265
|
// Bounded, because this blocks the control socket and the
|
|
262
266
|
// socket is serial: a read that runs long does not just
|
|
263
|
-
// return late, it holds up every command behind it.
|
|
264
|
-
//
|
|
265
|
-
//
|
|
266
|
-
//
|
|
267
|
-
//
|
|
268
|
-
|
|
267
|
+
// return late, it holds up every command behind it.
|
|
268
|
+
//
|
|
269
|
+
// The bound is "as long as the caller is willing to wait",
|
|
270
|
+
// not a number picked for feel. The client gives up at 30 s
|
|
271
|
+
// (`control.js`), so anything past that is lost either way,
|
|
272
|
+
// and stopping earlier only converts a slow-but-correct
|
|
273
|
+
// read into a failure. It was 12 s, chosen against a runner
|
|
274
|
+
// that once spent 28 s inside a *tree* read — a cost the
|
|
275
|
+
// attribute batching then removed — and a hosted runner
|
|
276
|
+
// where Vision has no GPU promptly failed a screen read
|
|
277
|
+
// with "text recognition did not finish within 12s". OCR is
|
|
278
|
+
// 100–400 ms on a developer's machine and evidently much
|
|
279
|
+
// slower on a shared one; guessing its ceiling was the
|
|
280
|
+
// mistake.
|
|
281
|
+
let timedOut = group.wait(timeout: .now() + .milliseconds(readBudgetMs)) == .timedOut
|
|
269
282
|
var (axTree, axError, axMs, ocrElements, ocrError, ocrMs) = reads.snapshot()
|
|
270
283
|
if timedOut {
|
|
284
|
+
let secs = readBudgetMs / 1000
|
|
271
285
|
if wantAx, axTree.nodes.isEmpty, axError == nil {
|
|
272
|
-
axError = "the accessibility read did not finish within
|
|
286
|
+
axError = "the accessibility read did not finish within \(secs)s"
|
|
273
287
|
}
|
|
274
288
|
if wantOcr, ocrElements.isEmpty, ocrError == nil {
|
|
275
|
-
ocrError = "text recognition did not finish within
|
|
289
|
+
ocrError = "text recognition did not finish within \(secs)s"
|
|
276
290
|
}
|
|
277
291
|
}
|
|
278
292
|
// OCR failing is fatal to a screen read in a way a missing
|
|
@@ -419,11 +433,11 @@ case "run":
|
|
|
419
433
|
latencies.append(Double(DispatchTime.now().uptimeNanoseconds - t0) / 1e6)
|
|
420
434
|
frames += 1
|
|
421
435
|
lastCapture = now
|
|
422
|
-
|
|
436
|
+
recovery.captureSucceeded()
|
|
423
437
|
} catch {
|
|
424
|
-
|
|
438
|
+
let due = recovery.captureFailed()
|
|
425
439
|
FileHandle.standardError.write(
|
|
426
|
-
"simframed: capture failed: \(error) (\(consecutiveFailures) in a row)\n".data(using: .utf8)!)
|
|
440
|
+
"simframed: capture failed: \(error) (\(recovery.consecutiveFailures) in a row)\n".data(using: .utf8)!)
|
|
427
441
|
// The display port can be torn down and rebuilt under a
|
|
428
442
|
// running daemon, and every read on the old descriptor
|
|
429
443
|
// returns nil from then on. Observed on a device that was
|
|
@@ -432,18 +446,16 @@ case "run":
|
|
|
432
446
|
// by restarting the daemon. Reporting a failure loudly is
|
|
433
447
|
// right; never recovering from it is not, so re-resolve the
|
|
434
448
|
// port and re-arm the damage callback.
|
|
435
|
-
if
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
}
|
|
449
|
+
if due {
|
|
450
|
+
switch recovery.reattach(platform: platform, onDamage: {
|
|
451
|
+
lock.lock(); dirty = true; lock.unlock()
|
|
452
|
+
}) {
|
|
453
|
+
case .success(let after):
|
|
441
454
|
lock.lock(); dirty = true; lock.unlock()
|
|
442
455
|
FileHandle.standardError.write(
|
|
443
|
-
"simframed: re-resolved the display port after \(
|
|
456
|
+
"simframed: re-resolved the display port after \(after) failed reads\n"
|
|
444
457
|
.data(using: .utf8)!)
|
|
445
|
-
|
|
446
|
-
} catch {
|
|
458
|
+
case .failure(let error):
|
|
447
459
|
FileHandle.standardError.write(
|
|
448
460
|
"simframed: could not re-resolve the display port: \(error)\n".data(using: .utf8)!)
|
|
449
461
|
}
|
|
@@ -268,3 +268,54 @@ final class AccessibilityElementTests: XCTestCase {
|
|
|
268
268
|
XCTAssertNil(element.json["identifier"], "an absent identifier is absent, not null")
|
|
269
269
|
}
|
|
270
270
|
}
|
|
271
|
+
|
|
272
|
+
/// The recovery path that only ever runs when nobody is watching.
|
|
273
|
+
///
|
|
274
|
+
/// A display teardown cannot be induced on demand, which is why this went
|
|
275
|
+
/// untested through two releases. The decision and the act can both be driven
|
|
276
|
+
/// with a stub, and that covers everything except the teardown itself.
|
|
277
|
+
final class CaptureRecoveryTests: XCTestCase {
|
|
278
|
+
func testAMomentaryHiccupDoesNotReattach() {
|
|
279
|
+
var recovery = CaptureRecovery(threshold: 6)
|
|
280
|
+
for _ in 0..<5 { XCTAssertFalse(recovery.captureFailed(), "five failures is a hiccup") }
|
|
281
|
+
XCTAssertEqual(recovery.consecutiveFailures, 5)
|
|
282
|
+
recovery.captureSucceeded()
|
|
283
|
+
XCTAssertEqual(recovery.consecutiveFailures, 0, "one good frame clears the run")
|
|
284
|
+
for _ in 0..<5 { XCTAssertFalse(recovery.captureFailed()) }
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
func testASustainedRunReattachesAndRearmsTheCallback() {
|
|
288
|
+
let platform = StubPlatform()
|
|
289
|
+
_ = try? platform.attach(udid: "STUB-1")
|
|
290
|
+
var recovery = CaptureRecovery(threshold: 3)
|
|
291
|
+
XCTAssertFalse(recovery.captureFailed())
|
|
292
|
+
XCTAssertFalse(recovery.captureFailed())
|
|
293
|
+
XCTAssertTrue(recovery.captureFailed(), "the third failure is due a re-resolve")
|
|
294
|
+
|
|
295
|
+
var damaged = false
|
|
296
|
+
let outcome = recovery.reattach(platform: platform) { damaged = true }
|
|
297
|
+
guard case .success(let after) = outcome else { return XCTFail("reattach should succeed on a live stub") }
|
|
298
|
+
XCTAssertEqual(after, 3, "it reports how many reads it lost")
|
|
299
|
+
XCTAssertEqual(platform.reattachCount, 1)
|
|
300
|
+
XCTAssertEqual(recovery.consecutiveFailures, 0)
|
|
301
|
+
|
|
302
|
+
// The callback half is the one that is easy to forget: a fresh
|
|
303
|
+
// descriptor with nothing registered on it gives a daemon that has
|
|
304
|
+
// recovered and will never notice another change.
|
|
305
|
+
platform.simulateChange()
|
|
306
|
+
XCTAssertTrue(damaged, "the damage callback was re-armed on the new descriptor")
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
func testAFailedReattachStaysDueRatherThanWaitingForAnotherSix() {
|
|
310
|
+
let platform = StubPlatform()
|
|
311
|
+
platform.failReattach = true
|
|
312
|
+
var recovery = CaptureRecovery(threshold: 2)
|
|
313
|
+
_ = recovery.captureFailed()
|
|
314
|
+
XCTAssertTrue(recovery.captureFailed())
|
|
315
|
+
guard case .failure = recovery.reattach(platform: platform, onDamage: {}) else {
|
|
316
|
+
return XCTFail("a stub told to fail should fail")
|
|
317
|
+
}
|
|
318
|
+
XCTAssertEqual(recovery.consecutiveFailures, 2, "the run is not cleared by an attempt that did not work")
|
|
319
|
+
XCTAssertTrue(recovery.captureFailed(), "so the next failure tries again immediately")
|
|
320
|
+
}
|
|
321
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "simframe",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.2",
|
|
4
4
|
"mcpName": "io.github.lvlrSajjad/simframe",
|
|
5
5
|
"description": "Always-warm iOS Simulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
|
|
6
6
|
"keywords": [
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
// Swift source or a new module here and it becomes required automatically.
|
|
16
16
|
import { execFileSync } from 'node:child_process';
|
|
17
17
|
import fs from 'node:fs';
|
|
18
|
+
import os from 'node:os';
|
|
18
19
|
import path from 'node:path';
|
|
19
20
|
import { fileURLToPath } from 'node:url';
|
|
20
21
|
|
|
@@ -83,8 +84,34 @@ function requiredFiles() {
|
|
|
83
84
|
}
|
|
84
85
|
|
|
85
86
|
const required = requiredFiles();
|
|
86
|
-
|
|
87
|
-
|
|
87
|
+
// Read the tarball, not npm's description of it.
|
|
88
|
+
//
|
|
89
|
+
// This used to parse `npm pack --dry-run --json` as `packed[0].files`, which is
|
|
90
|
+
// npm 10's shape. npm 12 reports something else, and the check died with
|
|
91
|
+
// "Cannot read properties of undefined (reading 'files')" — a check that exists
|
|
92
|
+
// to catch a silently broken package, itself silently broken by the tool it
|
|
93
|
+
// asks. It had only ever run under the npm bundled with Node 18–22; the release
|
|
94
|
+
// job runs a newer one.
|
|
95
|
+
//
|
|
96
|
+
// `tar -tzf` on the artifact npm actually produces has no shape to change, and
|
|
97
|
+
// it answers the stronger question: what is *in* the file people download.
|
|
98
|
+
const out = fs.mkdtempSync(path.join(os.tmpdir(), 'simframe-pack-'));
|
|
99
|
+
const name = execFileSync('npm', ['pack', '--silent', '--pack-destination', out], { cwd: ROOT, encoding: 'utf8' })
|
|
100
|
+
.trim()
|
|
101
|
+
.split('\n')
|
|
102
|
+
.pop();
|
|
103
|
+
const listing = execFileSync('tar', ['-tzf', path.join(out, name)], { encoding: 'utf8' });
|
|
104
|
+
fs.rmSync(out, { recursive: true, force: true });
|
|
105
|
+
const shipped = new Set(
|
|
106
|
+
listing
|
|
107
|
+
.split('\n')
|
|
108
|
+
.map((line) => line.trim())
|
|
109
|
+
.filter(Boolean)
|
|
110
|
+
// Every path in an npm tarball is prefixed `package/`, and directories
|
|
111
|
+
// arrive with a trailing slash.
|
|
112
|
+
.filter((line) => line.startsWith('package/') && !line.endsWith('/'))
|
|
113
|
+
.map((line) => line.slice('package/'.length)),
|
|
114
|
+
);
|
|
88
115
|
const missing = required.filter((f) => !shipped.has(f));
|
|
89
116
|
|
|
90
117
|
console.log(`${required.length} build inputs required, ${shipped.size} files in the tarball`);
|
package/scripts/ci-memory.mjs
CHANGED
|
@@ -41,6 +41,15 @@ const CONVERGE_PASSES = 3;
|
|
|
41
41
|
const BETWEEN_PASSES_MS = 1200;
|
|
42
42
|
|
|
43
43
|
let failures = 0;
|
|
44
|
+
/**
|
|
45
|
+
* The device is gone, as distinct from having blinked. See `jsonRetry`, which
|
|
46
|
+
* is the only thing that sets it: a dropped frame is retryable and this file
|
|
47
|
+
* already treats it that way, so calling the first one fatal would fight the
|
|
48
|
+
* retry rather than help it. Exhausted attempts are the difference between a
|
|
49
|
+
* blink and a death.
|
|
50
|
+
*/
|
|
51
|
+
let deviceDied = null;
|
|
52
|
+
|
|
44
53
|
function check(ok, label, detail = '') {
|
|
45
54
|
if (!ok) failures += 1;
|
|
46
55
|
console.log(`${ok ? 'ok ' : 'FAIL'} ${label}${detail ? ` — ${detail}` : ''}`);
|
|
@@ -65,6 +74,10 @@ async function cli(args, { expectFail = false, allowFail = false } = {}) {
|
|
|
65
74
|
if (err.unexpectedSuccess) throw err;
|
|
66
75
|
if (expectFail || allowFail) return `${err.stdout ?? ''}${err.stderr ?? ''}`;
|
|
67
76
|
const why = (err.stdout || err.stderr || err.message || '').trim();
|
|
77
|
+
// Noted here, not in `check`: by the time a failure reaches a check its
|
|
78
|
+
// detail has been truncated for legibility, and the first version of this
|
|
79
|
+
// guard looked for "did not produce a frame" in a string that had been cut
|
|
80
|
+
// to "simframe daemon di". The full text only exists at this boundary.
|
|
68
81
|
throw new Error(`simframe ${full.join(' ')} failed: ${why.slice(0, 400)}`);
|
|
69
82
|
}
|
|
70
83
|
}
|
|
@@ -97,6 +110,23 @@ async function jsonRetry(args, opts, attempts = 3) {
|
|
|
97
110
|
await new Promise((r) => setTimeout(r, 1500));
|
|
98
111
|
}
|
|
99
112
|
}
|
|
113
|
+
// Out of attempts on a capture error: the device is not blinking, it is gone.
|
|
114
|
+
//
|
|
115
|
+
// Diagnosed and exited here rather than flagged for a later `check` to
|
|
116
|
+
// notice, because most call sites do not wrap this — the throw escapes, the
|
|
117
|
+
// run dies on an unhandled rejection, and the operator gets a stack trace
|
|
118
|
+
// pointing at this file instead of a sentence about their simulator. Which is
|
|
119
|
+
// exactly what the first version of this did.
|
|
120
|
+
if (TRANSIENT.test(last?.message ?? '')) {
|
|
121
|
+
deviceDied = last.message;
|
|
122
|
+
console.error(`\nFAIL the device stopped producing frames, and did not come back after ${attempts} attempts:`);
|
|
123
|
+
console.error(` ${String(last.message).split('\n')[0]}`);
|
|
124
|
+
console.error('\nEverything after this point would be testing a dead simulator, so the run');
|
|
125
|
+
console.error('stops here. This is not a memory-layer failure — it is the device-state');
|
|
126
|
+
console.error('problem in docs/DEFERRED.md. A device restart is the only known cure;');
|
|
127
|
+
console.error('on a hosted runner it means a retry.');
|
|
128
|
+
process.exit(1);
|
|
129
|
+
}
|
|
100
130
|
throw last;
|
|
101
131
|
}
|
|
102
132
|
|
|
@@ -394,9 +424,14 @@ if (check(forced.saved?.ok === true, 'and --force saves it anyway', `${forced.sa
|
|
|
394
424
|
console.log('\n--- every command speaks JSON ---');
|
|
395
425
|
// The --json plumbing is per-command and hand-written, so one command quietly
|
|
396
426
|
// printing prose is exactly the kind of thing nothing else would catch.
|
|
427
|
+
// Through `jsonRetry` like everything else. This loop used to call `cli`
|
|
428
|
+
// directly, and it is the last section of a run that takes minutes — so a
|
|
429
|
+
// capture dropout here failed five checks about `--json` plumbing that was
|
|
430
|
+
// working perfectly, while every earlier section shrugged the same dropout off.
|
|
431
|
+
// The one place that did not retry was the one place most likely to need it.
|
|
397
432
|
for (const args of [['status'], ['state'], ['mark'], ['ui'], ['screens'], ['devices'], ['doctor'], ['flow', 'list'], ['recall']]) {
|
|
398
433
|
try {
|
|
399
|
-
const parsed =
|
|
434
|
+
const parsed = await jsonRetry([...args]);
|
|
400
435
|
check(parsed !== null && parsed !== undefined, `simframe ${args.join(' ')} --json`);
|
|
401
436
|
} catch (err) {
|
|
402
437
|
check(false, `simframe ${args.join(' ')} --json`, err.message.slice(0, 120));
|
package/src/fingerprint.js
CHANGED
|
@@ -11,6 +11,18 @@
|
|
|
11
11
|
import crypto from 'node:crypto';
|
|
12
12
|
import * as regions from './regions.js';
|
|
13
13
|
|
|
14
|
+
/**
|
|
15
|
+
* Bumped whenever the token rules change, and read by `graph.FINGERPRINT_VERSION`
|
|
16
|
+
* and `screenmap.MAP_VERSION` so stored hashes are discarded rather than
|
|
17
|
+
* compared against hashes computed by different rules. An old hash is a
|
|
18
|
+
* perfectly well-formed hash that never matches anything, which is the quietest
|
|
19
|
+
* kind of wrong.
|
|
20
|
+
*
|
|
21
|
+
* 2 — elements with no visible footprint, and containers holding two or more
|
|
22
|
+
* others, no longer enter identity: only one sensor can see either.
|
|
23
|
+
*/
|
|
24
|
+
export const TOKEN_RULES_VERSION = 2;
|
|
25
|
+
|
|
14
26
|
/** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
|
|
15
27
|
export const GRID = 24;
|
|
16
28
|
|
|
@@ -108,13 +120,31 @@ export function tokens(targets, screen) {
|
|
|
108
120
|
const keyboardTop = regions.detectKeyboardTop(targets, screen);
|
|
109
121
|
const groups = new Map();
|
|
110
122
|
|
|
123
|
+
// Identity is what the screen *is*, not which sensor happened to see it, so
|
|
124
|
+
// two things that only one sensor can produce must not enter it: an element
|
|
125
|
+
// with no visible footprint, and a container that exists to hold others.
|
|
126
|
+
// Pixels cannot see either, and the accessibility tree reports both.
|
|
127
|
+
const encloses = (frame) => targets.filter((o) => {
|
|
128
|
+
const f = o.frame;
|
|
129
|
+
if (!f || f === frame) return false;
|
|
130
|
+
const cx = f.x + (f.width ?? 0) / 2;
|
|
131
|
+
const cy = f.y + (f.height ?? 0) / 2;
|
|
132
|
+
return cx > frame.x && cx < frame.x + (frame.width ?? 0)
|
|
133
|
+
&& cy > frame.y && cy < frame.y + (frame.height ?? 0);
|
|
134
|
+
}).length;
|
|
135
|
+
|
|
111
136
|
for (const t of targets) {
|
|
112
137
|
const frame = t.frame ?? { x: t.x, y: t.y, width: 0, height: 0 };
|
|
113
138
|
// Off-screen elements are not part of what this screen looks like.
|
|
114
139
|
if (frame.y + (frame.height ?? 0) <= 0 || frame.y >= screen.height) continue;
|
|
140
|
+
// Nor is anything with no footprint to be seen.
|
|
141
|
+
if (!(frame.width > 0) || !(frame.height > 0)) continue;
|
|
115
142
|
const region = t.region ?? regions.regionFor(frame, screen, { keyboardTop });
|
|
116
143
|
if (region === 'status-bar') continue;
|
|
117
144
|
if (keyboardTop != null && frame.y >= keyboardTop) continue;
|
|
145
|
+
// A thing that holds two or more other things is scenery, and only the
|
|
146
|
+
// tree can see it. Its children are already in the fingerprint.
|
|
147
|
+
if (/group|other|generic/i.test(String(t.type ?? '')) && encloses(frame) >= 2) continue;
|
|
118
148
|
|
|
119
149
|
const role = roleOf(t);
|
|
120
150
|
// Group by what a thing IS and how big it is, not where it is. Repeated
|
|
@@ -157,6 +187,19 @@ export function tokens(targets, screen) {
|
|
|
157
187
|
}
|
|
158
188
|
|
|
159
189
|
export function hashTokens(list) {
|
|
190
|
+
// No tokens is not an identity. It is the absence of one.
|
|
191
|
+
//
|
|
192
|
+
// sha256 of the empty string is a constant, so every unreadable screen used
|
|
193
|
+
// to hash to `e3b0c442…` — one identity shared by a dark screen, a screen
|
|
194
|
+
// whose OCR failed, and a screen read mid-transition. Different screens
|
|
195
|
+
// collapsing onto a single hash is a wrong *merge*, which is worse than a
|
|
196
|
+
// missed match: a ref numbered on one screen resolved happily on another,
|
|
197
|
+
// and an edge learned on one predicted the other, with a verdict of `ok`.
|
|
198
|
+
//
|
|
199
|
+
// The pixel layout hash had exactly this degeneracy and got an
|
|
200
|
+
// `informative()` guard for it. The structural hash did not, and the guard
|
|
201
|
+
// could not have helped: a constant is perfectly informative-looking.
|
|
202
|
+
if (!list.length) return null;
|
|
160
203
|
return crypto.createHash('sha256').update(list.join('\n')).digest('hex').slice(0, 32);
|
|
161
204
|
}
|
|
162
205
|
|
|
@@ -174,7 +217,11 @@ export function fingerprint(targets, screen) {
|
|
|
174
217
|
* everything.
|
|
175
218
|
*/
|
|
176
219
|
export function similarity(a = [], b = []) {
|
|
177
|
-
|
|
220
|
+
// Two empty token sets are not a match, they are two absences of evidence.
|
|
221
|
+
// Returning 1 here made unreadability *self-confirming*: two consecutive
|
|
222
|
+
// unreadable reads "agreed", which promoted the non-identity to a confirmed
|
|
223
|
+
// screen and let it be written into memory and learned as an edge.
|
|
224
|
+
if (!a.length || !b.length) return 0;
|
|
178
225
|
const setA = new Set(a);
|
|
179
226
|
const setB = new Set(b);
|
|
180
227
|
let shared = 0;
|
package/src/graph.js
CHANGED
|
@@ -7,11 +7,28 @@
|
|
|
7
7
|
import fs from 'node:fs';
|
|
8
8
|
import path from 'node:path';
|
|
9
9
|
import { hashDistance } from './analyze.js';
|
|
10
|
+
import { informative } from './refs.js';
|
|
10
11
|
import * as fingerprint from './fingerprint.js';
|
|
11
12
|
import * as matching from './matching.js';
|
|
12
13
|
import * as store from './store.js';
|
|
13
14
|
|
|
14
|
-
const GRAPH_VERSION =
|
|
15
|
+
const GRAPH_VERSION = 3;
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Which fingerprint produced the hashes in these files.
|
|
19
|
+
*
|
|
20
|
+
* Separate from `GRAPH_VERSION` because it answers a different question: not
|
|
21
|
+
* "is this file shaped the way I expect" but "were these hashes computed by the
|
|
22
|
+
* same rules I am about to compare them with". A stored graph whose hashes came
|
|
23
|
+
* from an older fingerprint is not stale, it is *incomparable* — and the failure
|
|
24
|
+
* is silent, because an old hash is a perfectly well-formed hash that simply
|
|
25
|
+
* never matches anything.
|
|
26
|
+
*
|
|
27
|
+
* On a mismatch the graph is discarded and rebuilt, never translated. A rebuild
|
|
28
|
+
* costs a few hundred milliseconds per screen and happens once. A mis-merged
|
|
29
|
+
* graph costs a wrong tap, and costs it for as long as the file survives.
|
|
30
|
+
*/
|
|
31
|
+
export const FINGERPRINT_VERSION = fingerprint.TOKEN_RULES_VERSION;
|
|
15
32
|
/**
|
|
16
33
|
* Screens are matched by structural hash, exactly, and then by how alike their
|
|
17
34
|
* token sets are — which tolerates one optional element appearing (a badge, a
|
|
@@ -38,6 +55,14 @@ const GRAPH_VERSION = 2;
|
|
|
38
55
|
* Most revisits match on a hash outright and never reach this at all.
|
|
39
56
|
*/
|
|
40
57
|
export const SIMILARITY_THRESHOLD = 0.36;
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* How many of the 288 layout bits may differ and still be the same arrangement
|
|
61
|
+
* of light. The same number `screenmap` recalls maps by, and for the same
|
|
62
|
+
* reason: measured, a revisit is usually identical and different screens sit at
|
|
63
|
+
* 74 and above, so 20 is well inside the gap.
|
|
64
|
+
*/
|
|
65
|
+
export const SAME_SCREEN_LAYOUT_BITS = 20;
|
|
41
66
|
/**
|
|
42
67
|
* A screen with three async sections has a few settled structures, not endless
|
|
43
68
|
* ones. Capping this keeps a genuinely wrong merge bounded: if a node starts
|
|
@@ -105,11 +130,28 @@ function fingerprintsOf(node) {
|
|
|
105
130
|
function load(udid, screen) {
|
|
106
131
|
const key = typeof screen === 'string' ? { hash: screen, tokens: [] } : screen;
|
|
107
132
|
const entry = store.readJson(path.join(graphDir(udid), `${key.hash}.json`));
|
|
108
|
-
if (entry?.version === GRAPH_VERSION) return entry;
|
|
133
|
+
if (entry?.version === GRAPH_VERSION && entry?.fingerprintVersion === FINGERPRINT_VERSION) return entry;
|
|
109
134
|
// The hash may be a variant of a node filed under a different name.
|
|
110
135
|
const byVariant = allNodes(udid).find((n) => (n.variants ?? []).some((v) => v.hash === key.hash));
|
|
111
136
|
if (byVariant) return byVariant;
|
|
112
|
-
return {
|
|
137
|
+
return {
|
|
138
|
+
version: GRAPH_VERSION,
|
|
139
|
+
fingerprintVersion: FINGERPRINT_VERSION,
|
|
140
|
+
hash: key.hash,
|
|
141
|
+
tokens: key.tokens ?? [],
|
|
142
|
+
// What the pixels looked like here. Kept because it is the evidence that
|
|
143
|
+
// two structurally different readings are the same screen — see `record`.
|
|
144
|
+
layoutHash: key.layoutHash ?? null,
|
|
145
|
+
variants: [],
|
|
146
|
+
edges: [],
|
|
147
|
+
};
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/** Keep the pixel baseline current for a screen we are standing on. */
|
|
151
|
+
function noteLayout(node, reading) {
|
|
152
|
+
const now = typeof reading === 'string' ? null : reading?.layoutHash;
|
|
153
|
+
if (now && informative(now)) node.layoutHash = now;
|
|
154
|
+
return node;
|
|
113
155
|
}
|
|
114
156
|
|
|
115
157
|
function save(udid, node) {
|
|
@@ -124,7 +166,7 @@ export function allNodes(udid) {
|
|
|
124
166
|
.readdirSync(graphDir(udid))
|
|
125
167
|
.filter((f) => f.endsWith('.json'))
|
|
126
168
|
.map((f) => store.readJson(path.join(graphDir(udid), f)))
|
|
127
|
-
.filter((n) => n?.version === GRAPH_VERSION);
|
|
169
|
+
.filter((n) => n?.version === GRAPH_VERSION && n?.fingerprintVersion === FINGERPRINT_VERSION);
|
|
128
170
|
} catch {
|
|
129
171
|
return [];
|
|
130
172
|
}
|
|
@@ -243,6 +285,10 @@ export function record(udid, { from, action, to, kind }) {
|
|
|
243
285
|
// arrive as — that is what variants are for, and rewriting it here would let
|
|
244
286
|
// a node drift screen by screen into something it never was.
|
|
245
287
|
if (!node.tokens?.length && fromKey.tokens?.length) node.tokens = fromKey.tokens;
|
|
288
|
+
// The pixel baseline, on the other hand, *should* track: it is the evidence
|
|
289
|
+
// for "same arrangement of light as last time I stood here", and a stale one
|
|
290
|
+
// answers a question about a screen as it was weeks ago.
|
|
291
|
+
noteLayout(node, fromKey);
|
|
246
292
|
const to_ = toHash;
|
|
247
293
|
const signature = actionSignature(action);
|
|
248
294
|
const existing = node.edges.find((e) => e.action === signature);
|
|
@@ -260,8 +306,40 @@ export function record(udid, { from, action, to, kind }) {
|
|
|
260
306
|
const claimant = nearestScreen(udid, reading)?.node;
|
|
261
307
|
const target = load(udid, existing.to);
|
|
262
308
|
const unclaimed = !claimant || claimant.hash === target.hash;
|
|
263
|
-
|
|
309
|
+
// "Nothing we have stored claims this reading" is not evidence that the
|
|
310
|
+
// target grew a second face. It is equally consistent with this action
|
|
311
|
+
// being state-dependent and having gone somewhere genuinely new — and
|
|
312
|
+
// treating the two the same welds an unrelated screen onto a learned
|
|
313
|
+
// edge. Reproduced: a screen sharing *zero* tokens with the target got
|
|
314
|
+
// merged into it, after which landing there returned `ok` ("matches the
|
|
315
|
+
// outcome seen 3x before") and a flow kept walking, tapping real controls
|
|
316
|
+
// on a screen its plan never contained.
|
|
317
|
+
//
|
|
318
|
+
// So the reading has to positively look like the target before it is
|
|
319
|
+
// called a face of it. Unclaimed is a necessary condition, not a
|
|
320
|
+
// sufficient one.
|
|
321
|
+
// Two kinds of positive evidence that this reading is a face of the
|
|
322
|
+
// target, and either will do. What will not do is "nothing else claims
|
|
323
|
+
// it", which is an absence of evidence and used to be the whole test.
|
|
324
|
+
//
|
|
325
|
+
// * It looks like the target — the tokens overlap enough to be the same
|
|
326
|
+
// screen by the same measure used everywhere else.
|
|
327
|
+
// * It *looks like* the target on screen. The pixels are within the
|
|
328
|
+
// same-screen band of what was seen here before, and we arrived
|
|
329
|
+
// through an edge that has led here. Identity belongs to the graph as
|
|
330
|
+
// much as to the hash: the transition is evidence the fingerprint
|
|
331
|
+
// cannot supply, and it is exactly the evidence needed when one
|
|
332
|
+
// perception layer answered this time and not last time.
|
|
333
|
+
//
|
|
334
|
+
// That second route is what carries a screen whose structure genuinely
|
|
335
|
+
// differs between reads. Measured on four device-native screens, the
|
|
336
|
+
// accessibility tree and OCR agree on only 0.33–0.47 of a screen's
|
|
337
|
+
// tokens and never on its hash, and coarsening the vocabulary barely
|
|
338
|
+
// moved it — so this is not a residual case, it is the common one.
|
|
339
|
+
const looksLikeTarget = resembles(target, reading) || pixelsAgree(target, reading);
|
|
340
|
+
if (unclaimed && looksLikeTarget && target.hash !== to_ && reading.tokens?.length) {
|
|
264
341
|
addVariant(target, reading);
|
|
342
|
+
noteLayout(target, reading);
|
|
265
343
|
save(udid, target);
|
|
266
344
|
existing.count += 1;
|
|
267
345
|
existing.lastSeen = Date.now();
|
|
@@ -293,6 +371,39 @@ export function record(udid, { from, action, to, kind }) {
|
|
|
293
371
|
}
|
|
294
372
|
|
|
295
373
|
/** What this action did last time, if we have ever seen it here. */
|
|
374
|
+
/**
|
|
375
|
+
* Do the pixels say this is the same screen we have stood on here before?
|
|
376
|
+
*
|
|
377
|
+
* The layout hash is a poor answer to "which screen is this" on its own — that
|
|
378
|
+
* is why identity is structural — but it is a good answer to "is this the same
|
|
379
|
+
* arrangement of light", and combined with having arrived through a known edge
|
|
380
|
+
* it is the evidence that two structurally different readings are one screen.
|
|
381
|
+
*
|
|
382
|
+
* Guarded by `informative`, because a dark or uniform screen hashes to almost
|
|
383
|
+
* nothing and two of those are within any tolerance of each other while being
|
|
384
|
+
* evidence of nothing at all.
|
|
385
|
+
*/
|
|
386
|
+
function pixelsAgree(node, reading) {
|
|
387
|
+
const before = node?.layoutHash;
|
|
388
|
+
const now = reading?.layoutHash;
|
|
389
|
+
if (!before || !now || !informative(before) || !informative(now)) return false;
|
|
390
|
+
return hashDistance(before, now) <= SAME_SCREEN_LAYOUT_BITS;
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
/**
|
|
394
|
+
* Does this reading look like a face of this screen, rather than a different
|
|
395
|
+
* screen we happen not to have stored yet?
|
|
396
|
+
*
|
|
397
|
+
* Compared against every face the screen already wears, because a screen with
|
|
398
|
+
* two structures is exactly the case variants exist for and a new reading may
|
|
399
|
+
* resemble the second one rather than the first.
|
|
400
|
+
*/
|
|
401
|
+
function resembles(node, reading) {
|
|
402
|
+
if (!node || !reading?.tokens?.length) return false;
|
|
403
|
+
const faces = [node.tokens ?? [], ...(node.variants ?? []).map((v) => v.tokens ?? [])];
|
|
404
|
+
return faces.some((face) => face.length && fingerprint.similarity(face, reading.tokens) >= SIMILARITY_THRESHOLD);
|
|
405
|
+
}
|
|
406
|
+
|
|
296
407
|
export function predict(udid, from, action) {
|
|
297
408
|
const found = nearestScreen(udid, from);
|
|
298
409
|
if (!found) return null;
|
|
@@ -323,10 +434,27 @@ export function forget(udid) {
|
|
|
323
434
|
export function route(udid, fromHash, toHash, { maxDepth = 8 } = {}) {
|
|
324
435
|
const start = nearestScreen(udid, fromHash);
|
|
325
436
|
if (!start) return null;
|
|
326
|
-
|
|
327
|
-
if (goal(start.node.hash)) return [];
|
|
437
|
+
if (start.node.hash === toHash) return [];
|
|
328
438
|
|
|
329
|
-
|
|
439
|
+
// Variants have to resolve here the same way they resolve everywhere else.
|
|
440
|
+
//
|
|
441
|
+
// A screen may wear more than one structure, and an edge records whichever
|
|
442
|
+
// one it arrived on — so an edge whose `to` is a variant hash was a dead end
|
|
443
|
+
// in this search while `nearestScreen` was perfectly happy to say that hash
|
|
444
|
+
// *is* the node. The graph then had a route it could not find, `goto`
|
|
445
|
+
// answered `no-route` for somewhere it had been, and a flow that should have
|
|
446
|
+
// replayed from memory was re-explored instead. That is at least one
|
|
447
|
+
// mechanical cause of the convergence flakiness in DEFERRED.
|
|
448
|
+
const byHash = new Map();
|
|
449
|
+
for (const node of allNodes(udid)) {
|
|
450
|
+
byHash.set(node.hash, node);
|
|
451
|
+
for (const variant of node.variants ?? []) if (!byHash.has(variant.hash)) byHash.set(variant.hash, node);
|
|
452
|
+
}
|
|
453
|
+
// Canonical identity, so a variant and its screen are one place in the search
|
|
454
|
+
// rather than two, and reaching either counts as reaching the goal.
|
|
455
|
+
const canonical = (h) => byHash.get(h)?.hash ?? h;
|
|
456
|
+
const goalHash = canonical(toHash);
|
|
457
|
+
const reached = (h) => canonical(h) === goalHash;
|
|
330
458
|
const seen = new Set([start.node.hash]);
|
|
331
459
|
const queue = [{ hash: start.node.hash, path: [] }];
|
|
332
460
|
while (queue.length) {
|
|
@@ -334,11 +462,12 @@ export function route(udid, fromHash, toHash, { maxDepth = 8 } = {}) {
|
|
|
334
462
|
if (taken.length >= maxDepth) continue;
|
|
335
463
|
const node = byHash.get(hash);
|
|
336
464
|
for (const edge of node?.edges ?? []) {
|
|
337
|
-
|
|
465
|
+
const to = canonical(edge.to);
|
|
466
|
+
if (seen.has(to)) continue;
|
|
338
467
|
const next = [...taken, edge];
|
|
339
|
-
if (
|
|
340
|
-
seen.add(
|
|
341
|
-
queue.push({ hash:
|
|
468
|
+
if (reached(edge.to)) return next;
|
|
469
|
+
seen.add(to);
|
|
470
|
+
queue.push({ hash: to, path: next });
|
|
342
471
|
}
|
|
343
472
|
}
|
|
344
473
|
return null;
|
|
@@ -375,6 +504,18 @@ function sameScreen(udid, a, b) {
|
|
|
375
504
|
return Boolean(nodeA && nodeB && nodeA.hash === nodeB.hash);
|
|
376
505
|
}
|
|
377
506
|
|
|
507
|
+
/**
|
|
508
|
+
* How many times an edge must have been observed before a mismatch counts as a
|
|
509
|
+
* wrong turn rather than as "we do not know yet".
|
|
510
|
+
*
|
|
511
|
+
* Two, because the difference between one and two observations is the
|
|
512
|
+
* difference between a coincidence and a pattern, and the cost of being wrong
|
|
513
|
+
* is asymmetric: halting a correct run is visible and annoying, while
|
|
514
|
+
* continuing one extra step past a genuinely wrong turn is caught by the next
|
|
515
|
+
* step's own verdict.
|
|
516
|
+
*/
|
|
517
|
+
export const CONFIDENT_OBSERVATIONS = 2;
|
|
518
|
+
|
|
378
519
|
export function verdict({ udid, prediction, before, after, kind }) {
|
|
379
520
|
if (!before || !after) return { verdict: 'unverified', detail: 'no state to compare' };
|
|
380
521
|
const moved = before !== after;
|
|
@@ -387,6 +528,46 @@ export function verdict({ udid, prediction, before, after, kind }) {
|
|
|
387
528
|
return { verdict: 'no-visible-change', detail: `expected to reach a different screen (seen ${prediction.count}x)` };
|
|
388
529
|
}
|
|
389
530
|
if (!sameScreen(udid, prediction.to, after)) {
|
|
531
|
+
// One observation is not a prediction, and halting on it is what made a new
|
|
532
|
+
// user's *second* run worse than their first.
|
|
533
|
+
//
|
|
534
|
+
// The first run learns every edge at count 1 and cannot contradict itself,
|
|
535
|
+
// so it reports `unverified` throughout and completes. The second run then
|
|
536
|
+
// has an expectation for every step, and any screen whose identity wobbles
|
|
537
|
+
// — a read taken while the tree was still arriving, a screen with more than
|
|
538
|
+
// one legitimate structure — contradicts it and stops the run. Measured
|
|
539
|
+
// from the published package: run 1 halted at 3/10, runs 2 and 3 went
|
|
540
|
+
// 10/10. The halt was not protecting anyone from anything.
|
|
541
|
+
//
|
|
542
|
+
// So a single-observation miss is reported as what it actually is: we do
|
|
543
|
+
// not know yet. It must not be `unexpected-screen`, because that verdict is
|
|
544
|
+
// what halts a run and what "a run that reported a wrong turn never also
|
|
545
|
+
// reports success" is asserted over — and both of those should stay true.
|
|
546
|
+
// Once the same edge has been seen twice, a miss is a real wrong turn.
|
|
547
|
+
if (prediction.count < CONFIDENT_OBSERVATIONS) {
|
|
548
|
+
return {
|
|
549
|
+
verdict: 'unverified',
|
|
550
|
+
detail: `seen here once before and went somewhere else that time`
|
|
551
|
+
+ ` — one observation is not enough to call this a wrong turn`,
|
|
552
|
+
weakPrediction: { count: prediction.count, to: prediction.to },
|
|
553
|
+
};
|
|
554
|
+
}
|
|
555
|
+
// This action has already led somewhere different at least once, so its
|
|
556
|
+
// destination is not a fact about the screen — it is a distribution. An app
|
|
557
|
+
// relaunch that lands on restored state, a list whose first row depends on
|
|
558
|
+
// what happened last time: these genuinely have more than one outcome, and
|
|
559
|
+
// calling the second one a wrong turn is calling the world wrong.
|
|
560
|
+
//
|
|
561
|
+
// The graph has always counted this as `changedOutcomes` and nothing ever
|
|
562
|
+
// read it.
|
|
563
|
+
if (prediction.changedOutcomes > 0) {
|
|
564
|
+
return {
|
|
565
|
+
verdict: 'unverified',
|
|
566
|
+
detail: `this action has reached ${prediction.changedOutcomes + 1} different screens from here`
|
|
567
|
+
+ ` — its outcome is not predictable, so this is not a wrong turn`,
|
|
568
|
+
nondeterministic: { outcomes: prediction.changedOutcomes + 1, count: prediction.count },
|
|
569
|
+
};
|
|
570
|
+
}
|
|
390
571
|
return {
|
|
391
572
|
verdict: 'unexpected-screen',
|
|
392
573
|
detail: `expected the screen this action reached ${prediction.count}x before, and landed somewhere else`,
|
package/src/screenmap.js
CHANGED
|
@@ -13,9 +13,10 @@ import * as fingerprint from './fingerprint.js';
|
|
|
13
13
|
import * as input from './input.js';
|
|
14
14
|
import * as ocr from './ocr.js';
|
|
15
15
|
import * as regions from './regions.js';
|
|
16
|
+
import { informative } from './refs.js';
|
|
16
17
|
import * as store from './store.js';
|
|
17
18
|
|
|
18
|
-
const MAP_VERSION =
|
|
19
|
+
const MAP_VERSION = 7; // footprintless elements and containers no longer enter identity
|
|
19
20
|
|
|
20
21
|
function mapDir(udid) {
|
|
21
22
|
return path.join(store.deviceDir(udid), 'screens');
|
|
@@ -60,6 +61,12 @@ function loadAll(udid) {
|
|
|
60
61
|
*/
|
|
61
62
|
export function recallNearest(udid, layoutHash, { tolerance = DEFAULT_TOLERANCE } = {}) {
|
|
62
63
|
if (!layoutHash) return null;
|
|
64
|
+
// A hash of almost no set bits is a dark or uniform screen, and two of them
|
|
65
|
+
// are within any tolerance of each other while being evidence of nothing.
|
|
66
|
+
// `refs.js` documents this and guards for it; this function fed that one's
|
|
67
|
+
// `screenKnown` and `structuralHash` inputs without a guard of its own, so a
|
|
68
|
+
// near-uniform screen could hand back a different screen's element map.
|
|
69
|
+
if (!informative(layoutHash)) return null;
|
|
63
70
|
let best = null;
|
|
64
71
|
let bestDistance = Infinity;
|
|
65
72
|
for (const entry of loadAll(udid)) {
|