simframe 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +181 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +72 -4
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/SimframeCore/CaptureRecovery.swift +19 -0
- package/native/simframed/Sources/simframed/main.swift +33 -3
- package/native/simframed/Tests/SimframeCoreTests/HashingTests.swift +41 -0
- package/native/supervise.swift +216 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +9 -0
- package/scripts/ci-integration-local.sh +79 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/collect-rulings.mjs +312 -0
- package/scripts/eval-fingerprint.mjs +100 -23
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/scripts/probe-network.mjs +118 -0
- package/scripts/soak-capture.mjs +72 -0
- package/skills/simframe/SKILL.md +257 -7
- package/src/actions.js +1791 -44
- package/src/cli.js +216 -14
- package/src/control.js +1 -0
- package/src/fingerprint.js +43 -1
- package/src/graph.js +136 -7
- package/src/index.js +252 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +161 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +333 -32
- package/src/metrics.js +148 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +25 -1
- package/src/platform/index.js +11 -1
- package/src/platform/ios.js +25 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +215 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +161 -0
- package/src/view.js +375 -11
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
|
@@ -105,6 +105,10 @@ extension StubPlatform {
|
|
|
105
105
|
}
|
|
106
106
|
public func type(_ text: String) throws { recorded.append("type(\(text))") }
|
|
107
107
|
public func paste(_ text: String) throws { recorded.append("paste(\(text))") }
|
|
108
|
+
public func pressKey(usage: UInt32, modifiers: [UInt32]) throws {
|
|
109
|
+
throw PrivateAPIError.frameworksUnavailable("stub platform")
|
|
110
|
+
}
|
|
111
|
+
|
|
108
112
|
public func press(_ button: HardwareButton) throws { recorded.append("press(\(button.rawValue))") }
|
|
109
113
|
public func longPress(at point: CGPoint, durationMs: Double) throws {
|
|
110
114
|
recorded.append("longPress(\(Int(point.x)),\(Int(point.y)),\(Int(durationMs)))")
|
|
@@ -75,6 +75,25 @@ public struct CaptureRecovery {
|
|
|
75
75
|
reattaches >= Self.stalledAfterReattaches && rebinds < Self.maxRebinds
|
|
76
76
|
}
|
|
77
77
|
|
|
78
|
+
/// Both rungs of the ladder have been tried and no frame has arrived since.
|
|
79
|
+
///
|
|
80
|
+
/// What happens after the ladder runs out was left implicit, and the
|
|
81
|
+
/// implicit answer was "go back to the bottom rung forever". Measured from a
|
|
82
|
+
/// real wedge: **670 re-resolves against 42 rebinds** in one log. Once
|
|
83
|
+
/// `rebinds` hits `maxRebinds` — and it only resets on a real frame —
|
|
84
|
+
/// `needsRebind` is false for good, so every sixth failure re-resolved a
|
|
85
|
+
/// port that two rebinds had already proved was not the problem, at 600ms a
|
|
86
|
+
/// go, each time logging a message that calls the condition "usually
|
|
87
|
+
/// transient".
|
|
88
|
+
///
|
|
89
|
+
/// That is why a wedged device reads as the tool hanging rather than the
|
|
90
|
+
/// tool reporting. The state is unchanged in spirit — the daemon still only
|
|
91
|
+
/// reports, and restarting the device stays the operator's call — but it can
|
|
92
|
+
/// stop pretending it has something left to try.
|
|
93
|
+
public var recoveryExhausted: Bool {
|
|
94
|
+
rebinds >= Self.maxRebinds && reattaches >= Self.stalledAfterReattaches
|
|
95
|
+
}
|
|
96
|
+
|
|
78
97
|
public mutating func captureSucceeded() {
|
|
79
98
|
consecutiveFailures = 0
|
|
80
99
|
// A real frame is the only evidence that health is back. Resetting this
|
|
@@ -122,7 +122,7 @@ case "input":
|
|
|
122
122
|
print(" \(device.name): \(device.pixelWidth)x\(device.pixelHeight)px @\(device.scale)x = \(device.pointWidth)x\(device.pointHeight)pt")
|
|
123
123
|
} catch { fail("\(error)") }
|
|
124
124
|
|
|
125
|
-
case "tap", "swipe", "type", "paste", "press":
|
|
125
|
+
case "tap", "swipe", "type", "paste", "press", "key":
|
|
126
126
|
do {
|
|
127
127
|
_ = try platform.attach(udid: flag("udid"))
|
|
128
128
|
let positional = args.filter { !$0.hasPrefix("--") }.dropFirst()
|
|
@@ -141,6 +141,11 @@ case "tap", "swipe", "type", "paste", "press":
|
|
|
141
141
|
case "type":
|
|
142
142
|
guard !positional.isEmpty else { fail("usage: simframed type <text>") }
|
|
143
143
|
try platform.type(positional.joined(separator: " "))
|
|
144
|
+
case "key":
|
|
145
|
+
guard let u = positional.first, let usage = UInt32(u) else {
|
|
146
|
+
fail("usage: simframed key <hid-usage-code>")
|
|
147
|
+
}
|
|
148
|
+
try platform.pressKey(usage: usage, modifiers: [])
|
|
144
149
|
case "paste":
|
|
145
150
|
guard !positional.isEmpty else { fail("usage: simframed paste <text>") }
|
|
146
151
|
try platform.paste(positional.joined(separator: " "))
|
|
@@ -187,6 +192,8 @@ case "run":
|
|
|
187
192
|
var dirty = true
|
|
188
193
|
var recovery = CaptureRecovery()
|
|
189
194
|
var stalledSince: Double?
|
|
195
|
+
/// Said once per stall episode, not once per failed read.
|
|
196
|
+
var announcedExhausted = false
|
|
190
197
|
var lastCapture = 0.0
|
|
191
198
|
var frames = 0
|
|
192
199
|
var lastReport = Date().timeIntervalSince1970
|
|
@@ -404,6 +411,13 @@ case "run":
|
|
|
404
411
|
}
|
|
405
412
|
try platform.press(button)
|
|
406
413
|
return done()
|
|
414
|
+
case "key":
|
|
415
|
+
guard let usage = request["usage"] as? NSNumber else {
|
|
416
|
+
return ["ok": false, "error": "key needs a HID usage code"]
|
|
417
|
+
}
|
|
418
|
+
let mods = (request["modifiers"] as? [NSNumber])?.map { $0.uint32Value } ?? []
|
|
419
|
+
try platform.pressKey(usage: usage.uint32Value, modifiers: mods)
|
|
420
|
+
return done()
|
|
407
421
|
case "resetInput":
|
|
408
422
|
try platform.resetInput()
|
|
409
423
|
FileHandle.standardError.write("simframed: HID session reset on request\n".data(using: .utf8)!)
|
|
@@ -483,6 +497,7 @@ case "run":
|
|
|
483
497
|
// back on its own, which nobody would otherwise know.
|
|
484
498
|
try? store.writeCaptureHealth(nil)
|
|
485
499
|
stalledSince = nil
|
|
500
|
+
announcedExhausted = false
|
|
486
501
|
FileHandle.standardError.write(
|
|
487
502
|
"simframed: capture recovered on its own\n".data(using: .utf8)!)
|
|
488
503
|
}
|
|
@@ -498,7 +513,19 @@ case "run":
|
|
|
498
513
|
// by restarting the daemon. Reporting a failure loudly is
|
|
499
514
|
// right; never recovering from it is not, so re-resolve the
|
|
500
515
|
// port and re-arm the damage callback.
|
|
501
|
-
|
|
516
|
+
// Nothing left to try is a thing to say once, not a
|
|
517
|
+
// rung to keep pulling. See `recoveryExhausted`.
|
|
518
|
+
if due && recovery.recoveryExhausted {
|
|
519
|
+
if !announcedExhausted {
|
|
520
|
+
announcedExhausted = true
|
|
521
|
+
FileHandle.standardError.write(
|
|
522
|
+
("simframed: capture is wedged and both recoveries are spent"
|
|
523
|
+
+ " (\(recovery.reattaches) port re-resolves, \(recovery.rebinds) device rebinds,"
|
|
524
|
+
+ " no frame since). This needs the device restarted —"
|
|
525
|
+
+ " `simframe revive --device=<udid>`. Backing off until a frame arrives.\n")
|
|
526
|
+
.data(using: .utf8)!)
|
|
527
|
+
}
|
|
528
|
+
} else if due {
|
|
502
529
|
let onDamage = { lock.lock(); dirty = true; lock.unlock() }
|
|
503
530
|
// Escalate rather than repeat. Two successful re-resolves
|
|
504
531
|
// with no frame between them means the port was never the
|
|
@@ -540,7 +567,10 @@ case "run":
|
|
|
540
567
|
"reason": "\(error)",
|
|
541
568
|
])
|
|
542
569
|
}
|
|
543
|
-
|
|
570
|
+
// Back off once there is nothing left to attempt. Half a
|
|
571
|
+
// second forever on a dead display is pure heat, and the
|
|
572
|
+
// log it produced buried everything else in the file.
|
|
573
|
+
Thread.sleep(forTimeInterval: recovery.recoveryExhausted ? 5.0 : 0.5)
|
|
544
574
|
}
|
|
545
575
|
}
|
|
546
576
|
}
|
|
@@ -397,4 +397,45 @@ final class CaptureRecoveryTests: XCTestCase {
|
|
|
397
397
|
XCTAssertEqual(recovery.consecutiveFailures, 2, "the run is not cleared by an attempt that did not work")
|
|
398
398
|
XCTAssertTrue(recovery.captureFailed(), "so the next failure tries again immediately")
|
|
399
399
|
}
|
|
400
|
+
|
|
401
|
+
/// What happens when the ladder runs out, which was previously implicit —
|
|
402
|
+
/// and the implicit answer was "start again at the bottom, forever".
|
|
403
|
+
func testAnExhaustedLadderStopsPretendingItHasSomethingLeft() {
|
|
404
|
+
let platform = StubPlatform()
|
|
405
|
+
var recovery = CaptureRecovery(threshold: 1)
|
|
406
|
+
XCTAssertFalse(recovery.recoveryExhausted, "nothing has been tried yet")
|
|
407
|
+
|
|
408
|
+
// Two re-resolves that each succeed and change nothing: this is the
|
|
409
|
+
// pathology, not a hypothetical. The port hands back a fresh descriptor
|
|
410
|
+
// and every read still fails.
|
|
411
|
+
for _ in 0..<CaptureRecovery.stalledAfterReattaches {
|
|
412
|
+
_ = recovery.captureFailed()
|
|
413
|
+
guard case .success = recovery.reattach(platform: platform, onDamage: {}) else {
|
|
414
|
+
return XCTFail("the stub re-resolves")
|
|
415
|
+
}
|
|
416
|
+
}
|
|
417
|
+
XCTAssertTrue(recovery.needsRebind, "so the escalation is due")
|
|
418
|
+
XCTAssertFalse(recovery.recoveryExhausted, "but the ladder has a second rung")
|
|
419
|
+
|
|
420
|
+
for _ in 0..<CaptureRecovery.maxRebinds {
|
|
421
|
+
_ = recovery.captureFailed()
|
|
422
|
+
guard case .success = recovery.rebind(platform: platform, udid: "UDID", onDamage: {}) else {
|
|
423
|
+
return XCTFail("the stub rebinds")
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
XCTAssertFalse(recovery.needsRebind, "the rebinds are spent")
|
|
427
|
+
XCTAssertTrue(recovery.recoveryExhausted, "and so is the ladder")
|
|
428
|
+
|
|
429
|
+
// Measured from a real wedge: 670 port re-resolves against 42 rebinds in
|
|
430
|
+
// one log, because this state fell back to the bottom rung every sixth
|
|
431
|
+
// failure at 600ms a go, each time logging "usually transient".
|
|
432
|
+
_ = recovery.captureFailed()
|
|
433
|
+
XCTAssertTrue(recovery.recoveryExhausted, "more failures do not restore an option")
|
|
434
|
+
|
|
435
|
+
// A real frame is the only thing that resets it, exactly as for the rest
|
|
436
|
+
// of this machine — a re-resolve must never look like recovery.
|
|
437
|
+
recovery.captureSucceeded()
|
|
438
|
+
XCTAssertFalse(recovery.recoveryExhausted, "a frame is the only evidence health is back")
|
|
439
|
+
XCTAssertFalse(recovery.isStalled)
|
|
440
|
+
}
|
|
400
441
|
}
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
// The local supervisor, and it may say exactly three things.
|
|
2
|
+
//
|
|
3
|
+
// Claude plans. The deterministic executor in src/actions.js runs the plan and
|
|
4
|
+
// verifies each step — round 6's best result was 18 steps in one call, and code
|
|
5
|
+
// is a better executor than a model: faster, exact, auditable, and it cannot
|
|
6
|
+
// hallucinate a step. What the executor has never had is judgement at the
|
|
7
|
+
// moment a step fails, so a batch died and a round trip was spent on a decision
|
|
8
|
+
// that was usually obvious.
|
|
9
|
+
//
|
|
10
|
+
// This answers that moment and nothing else:
|
|
11
|
+
//
|
|
12
|
+
// in {"goal":"...","step":"tap REVIEW","expected":"...","failure":"...","screen":["…labels…"],"stillMs":120}
|
|
13
|
+
// out {"decision":"wait","ms":410}
|
|
14
|
+
//
|
|
15
|
+
// `wait`, `retry`, `stop`. It cannot invent a step, skip one, substitute a
|
|
16
|
+
// target, or continue past an unexpected screen — not because a threshold
|
|
17
|
+
// forbids it but because those are not words it can say. That constraint is the
|
|
18
|
+
// safety property. `seek` was given latitude over *what* to open and pressed
|
|
19
|
+
// "YES, THIS FIXED MY PROBLEM" in a live app; a component whose whole answer
|
|
20
|
+
// space is three words cannot do that whatever it believes.
|
|
21
|
+
//
|
|
22
|
+
// Unavailable is a normal answer. It says so and exits, `doctor` reports
|
|
23
|
+
// `supervisor: none`, and the executor behaves exactly as it does today.
|
|
24
|
+
import Foundation
|
|
25
|
+
#if canImport(FoundationModels)
|
|
26
|
+
import FoundationModels
|
|
27
|
+
|
|
28
|
+
@available(macOS 26.0, *)
|
|
29
|
+
@Generable
|
|
30
|
+
enum Decision: String {
|
|
31
|
+
case wait
|
|
32
|
+
case retry
|
|
33
|
+
case stop
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
@available(macOS 26.0, *)
|
|
37
|
+
@Generable
|
|
38
|
+
struct Judgement {
|
|
39
|
+
@Guide(description: "wait if the screen is still arriving, retry if the same step should be attempted again, stop if nothing further can work")
|
|
40
|
+
var decision: Decision
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
struct Situation: Decodable {
|
|
44
|
+
let goal: String?
|
|
45
|
+
let step: String
|
|
46
|
+
let expected: String?
|
|
47
|
+
let failure: String
|
|
48
|
+
let screen: [String]?
|
|
49
|
+
let stillMs: Int?
|
|
50
|
+
let note: String?
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
func emit(_ object: [String: Any]) {
|
|
54
|
+
guard let data = try? JSONSerialization.data(withJSONObject: object),
|
|
55
|
+
let line = String(data: data, encoding: .utf8) else { return }
|
|
56
|
+
print(line)
|
|
57
|
+
fflush(stdout)
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
@available(macOS 26.0, *)
|
|
61
|
+
func serve() async {
|
|
62
|
+
switch SystemLanguageModel.default.availability {
|
|
63
|
+
case .available: break
|
|
64
|
+
case .unavailable(let reason):
|
|
65
|
+
emit(["unavailable": "\(reason)"])
|
|
66
|
+
return
|
|
67
|
+
@unknown default:
|
|
68
|
+
emit(["unavailable": "unknown availability"])
|
|
69
|
+
return
|
|
70
|
+
}
|
|
71
|
+
// Instructions, reused; the *session* is not.
|
|
72
|
+
//
|
|
73
|
+
// One session reused across requests is what made the supervisor go silent
|
|
74
|
+
// in the field: `LanguageModelSession` accumulates its transcript, so after
|
|
75
|
+
// seven real failures it hit `exceededContextWindowSize` — 4,441 tokens
|
|
76
|
+
// against a 4,096 maximum — and every request after that errored. The tester
|
|
77
|
+
// saw three rulings and then nothing for twenty supervised calls, while
|
|
78
|
+
// `doctor` in a separate process kept reporting the model available. It
|
|
79
|
+
// degraded before it broke, too: latency climbed 987ms to 1,890ms as the
|
|
80
|
+
// transcript grew, and the reasons collapsed into boilerplate repeated
|
|
81
|
+
// verbatim across unrelated failures.
|
|
82
|
+
//
|
|
83
|
+
// Each judgement is independent by nature — a failed step, what was
|
|
84
|
+
// expected, what is on screen — so there is nothing for a transcript to
|
|
85
|
+
// carry, and carrying it was pure cost even before it was fatal. The process
|
|
86
|
+
// stays warm, which is where the model load is paid; only the conversation
|
|
87
|
+
// is fresh.
|
|
88
|
+
let instructions = """
|
|
89
|
+
You supervise a UI test that is running a plan someone else wrote. A step \
|
|
90
|
+
has just failed. Decide one of three things and nothing else.
|
|
91
|
+
|
|
92
|
+
wait — the screen is still arriving: a spinner, a list that has not \
|
|
93
|
+
rendered, a count header with no rows, a transition in progress. Waiting \
|
|
94
|
+
would let the same step succeed.
|
|
95
|
+
retry — the step is sound and the moment was wrong: something moved under \
|
|
96
|
+
it, focus was lost, a stale reading. Repeating it would work.
|
|
97
|
+
stop — nothing further in the plan can work: the app is somewhere else, a \
|
|
98
|
+
required control is absent or disabled for a reason, or the screen has not \
|
|
99
|
+
responded at all.
|
|
100
|
+
|
|
101
|
+
Weigh the evidence you are given, in this order.
|
|
102
|
+
|
|
103
|
+
The plan's own guidance about this app comes FIRST and outranks everything \
|
|
104
|
+
below. It is the only knowledge of this app you have, and it was written by \
|
|
105
|
+
someone who has seen it. If it says lists arrive late, then a row that is \
|
|
106
|
+
not there yet means wait — even on a screen that has gone completely still, \
|
|
107
|
+
because a screen waiting on a network call is perfectly still and perfectly \
|
|
108
|
+
empty.
|
|
109
|
+
|
|
110
|
+
If a note says the screen is still filling in — a spinner, a count header \
|
|
111
|
+
with too few rows, a transition — that is decisive: answer wait.
|
|
112
|
+
If the screen has been still for under about 1500ms, content may still be \
|
|
113
|
+
arriving: prefer wait.
|
|
114
|
+
If the visible labels belong to a different part of the app than the step \
|
|
115
|
+
implies, answer stop.
|
|
116
|
+
If a control the step names is present but refused, answer stop.
|
|
117
|
+
Otherwise, if the step looks like it simply caught a bad moment, answer \
|
|
118
|
+
retry.
|
|
119
|
+
|
|
120
|
+
Never suggest a different step, a different target, or skipping ahead. You \
|
|
121
|
+
are not being asked what to do, only whether this can proceed. Answer with \
|
|
122
|
+
the decision alone.
|
|
123
|
+
"""
|
|
124
|
+
// Warm up front so the first real judgement does not pay model load —
|
|
125
|
+
// measured at ~880ms against ~600ms warm.
|
|
126
|
+
//
|
|
127
|
+
// `prewarm()` is the supported way and this used to be a throwaway
|
|
128
|
+
// `respond`, which loaded the same weights the hard way and paid a whole
|
|
129
|
+
// generation to do it. Warming a session we then discard works because
|
|
130
|
+
// residency is process- and system-managed: the expensive part outlives the
|
|
131
|
+
// session, which is also why fresh-session-per-judgement is affordable.
|
|
132
|
+
LanguageModelSession(instructions: instructions).prewarm()
|
|
133
|
+
// The window, reported rather than assumed. 4,096 has been a documented
|
|
134
|
+
// constant we repeated; since 26.4 it is queryable, so it is now read from
|
|
135
|
+
// the model and handed to the caller, who prints it in `doctor`.
|
|
136
|
+
emit(["ready": true, "contextSize": SystemLanguageModel.default.contextSize])
|
|
137
|
+
while let line = readLine(strippingNewline: true) {
|
|
138
|
+
if line.isEmpty { continue }
|
|
139
|
+
guard let data = line.data(using: .utf8),
|
|
140
|
+
let s = try? JSONDecoder().decode(Situation.self, from: data) else {
|
|
141
|
+
emit(["error": "could not parse that line"])
|
|
142
|
+
continue
|
|
143
|
+
}
|
|
144
|
+
// Capped here as well as in the caller. A single realistic failure
|
|
145
|
+
// message carries the whole "Visible: …" list, and one request built
|
|
146
|
+
// from raw input measured 4,266 tokens against a 4,096 window — so a
|
|
147
|
+
// request can exceed the context on its own, with no transcript
|
|
148
|
+
// involved at all. Truncation belongs where the limit is.
|
|
149
|
+
let clip = { (t: String, n: Int) in t.count > n ? String(t.prefix(n)) + "…" : t }
|
|
150
|
+
var prompt = "Step: \(clip(s.step, 120))\nIt failed with: \(clip(s.failure, 220))"
|
|
151
|
+
if let goal = s.goal, !goal.isEmpty { prompt += "\nThe plan's guidance about this app: \(clip(goal, 300))" }
|
|
152
|
+
if let expected = s.expected, !expected.isEmpty { prompt += "\nExpected: \(clip(expected, 200))" }
|
|
153
|
+
if let ms = s.stillMs { prompt += "\nThe screen has been still for \(ms)ms" }
|
|
154
|
+
if let note = s.note, !note.isEmpty { prompt += "\nPerception note: \(clip(note, 160))" }
|
|
155
|
+
if let screen = s.screen, !screen.isEmpty {
|
|
156
|
+
prompt += "\nOn screen now: \(screen.prefix(14).map { clip($0, 28) }.joined(separator: ", "))"
|
|
157
|
+
}
|
|
158
|
+
let started = Date()
|
|
159
|
+
do {
|
|
160
|
+
let session = LanguageModelSession(instructions: instructions)
|
|
161
|
+
let out = try await session.respond(to: prompt, generating: Judgement.self)
|
|
162
|
+
// No reason field, deliberately. Asked for one it confabulated in
|
|
163
|
+
// every observed run: a correct `stop` justified as "screen is
|
|
164
|
+
// elsewhere" when the screen was exactly where the plan expected,
|
|
165
|
+
// and reasons repeated verbatim across unrelated failures. The
|
|
166
|
+
// reporter's judgement, and it is right — *"a right answer with a
|
|
167
|
+
// fabricated justification teaches me to distrust the
|
|
168
|
+
// justification, which is most of the value of it explaining
|
|
169
|
+
// itself. No reason at all would be better than a confident wrong
|
|
170
|
+
// one."* What the caller gets instead is which rule or which model
|
|
171
|
+
// answered, which is true by construction.
|
|
172
|
+
emit([
|
|
173
|
+
"decision": out.content.decision.rawValue,
|
|
174
|
+
"ms": Int(Date().timeIntervalSince(started) * 1000),
|
|
175
|
+
])
|
|
176
|
+
} catch let err as LanguageModelSession.GenerationError {
|
|
177
|
+
// Named, not merely stringified, because the cases mean different
|
|
178
|
+
// things to whoever reads the log — and because a string match on
|
|
179
|
+
// an error message is a classification that silently becomes
|
|
180
|
+
// "unknown" the day Apple rewords it.
|
|
181
|
+
//
|
|
182
|
+
// Deliberately NOT a retry policy. The research that asked for this
|
|
183
|
+
// warned that a blanket retry burns battery on the permanent cases,
|
|
184
|
+
// and checking found we never retry at all: `ask` resolves null on
|
|
185
|
+
// any error and `judge` refuses anything outside the three words. So
|
|
186
|
+
// what these buy is a diagnosis, which is what was actually missing
|
|
187
|
+
// — every failure reached the log as "the supervisor did not answer".
|
|
188
|
+
let kind: String
|
|
189
|
+
switch err {
|
|
190
|
+
// Our own bug if it appears: the caller clips every field and the
|
|
191
|
+
// worst case those caps allow measures 1,918 tokens of 4,096.
|
|
192
|
+
case .exceededContextWindowSize: kind = "context-window"
|
|
193
|
+
// Permanent for this input. Retrying cannot help and it says so.
|
|
194
|
+
case .guardrailViolation: kind = "guardrail"
|
|
195
|
+
case .unsupportedLanguageOrLocale: kind = "locale"
|
|
196
|
+
// A session takes one request at a time. The line protocol here
|
|
197
|
+
// serialises them and each gets a fresh session, so this should be
|
|
198
|
+
// unreachable — worth knowing loudly if it ever is not.
|
|
199
|
+
case .rateLimited: kind = "rate-limited"
|
|
200
|
+
default: kind = "generation"
|
|
201
|
+
}
|
|
202
|
+
emit(["error": "\(err)", "kind": kind])
|
|
203
|
+
} catch {
|
|
204
|
+
emit(["error": "\(error)", "kind": "unknown"])
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
if #available(macOS 26.0, *) {
|
|
210
|
+
await serve()
|
|
211
|
+
} else {
|
|
212
|
+
emit(["unavailable": "the on-device model needs macOS 26 or newer"])
|
|
213
|
+
}
|
|
214
|
+
#else
|
|
215
|
+
print("{\"unavailable\":\"this toolchain cannot import FoundationModels\"}")
|
|
216
|
+
#endif
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "simframe",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.12.0",
|
|
4
4
|
"mcpName": "io.github.lvlrSajjad/simframe",
|
|
5
5
|
"description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
|
|
6
6
|
"keywords": [
|
|
@@ -42,8 +42,11 @@
|
|
|
42
42
|
},
|
|
43
43
|
"files": [
|
|
44
44
|
"src",
|
|
45
|
+
"data",
|
|
45
46
|
"flows",
|
|
46
47
|
"native/ocr.swift",
|
|
48
|
+
"native/rank.swift",
|
|
49
|
+
"native/supervise.swift",
|
|
47
50
|
"native/simframed/Package.swift",
|
|
48
51
|
"native/simframed/Sources",
|
|
49
52
|
"native/simframed/Tests",
|
|
@@ -60,8 +60,28 @@ function requiredFiles() {
|
|
|
60
60
|
throw new Error('no Swift test sources found — has the layout moved? This check would silently pass.');
|
|
61
61
|
}
|
|
62
62
|
|
|
63
|
-
//
|
|
64
|
-
|
|
63
|
+
// Every standalone Swift helper compiled on demand from `src/`.
|
|
64
|
+
//
|
|
65
|
+
// This listed `native/ocr.swift` by name, which is why it did not catch the
|
|
66
|
+
// next two: `rank.swift` and `supervise.swift` were absent from `files`, so
|
|
67
|
+
// the local planner and the local supervisor would have been silently
|
|
68
|
+
// unavailable for every installed user — the exact failure this whole check
|
|
69
|
+
// exists to prevent, reproduced because the check named one file instead of
|
|
70
|
+
// stating the rule.
|
|
71
|
+
//
|
|
72
|
+
// The rule is: if a module under `src/` compiles it, it has to ship.
|
|
73
|
+
const helpers = walk(path.join(ROOT, 'native'), (p) => p.endsWith('.swift'))
|
|
74
|
+
.filter((p) => !p.includes(`${path.sep}simframed${path.sep}`) && !p.includes(`${path.sep}spikes${path.sep}`));
|
|
75
|
+
if (!helpers.length) {
|
|
76
|
+
throw new Error('no standalone Swift helper found under native/ — has the layout moved? This check would silently pass.');
|
|
77
|
+
}
|
|
78
|
+
const sources = walk(path.join(ROOT, 'src'), (p) => p.endsWith('.js'))
|
|
79
|
+
.map((f) => fs.readFileSync(f, 'utf8')).join('\n');
|
|
80
|
+
for (const helper of helpers) {
|
|
81
|
+
const name = path.basename(helper);
|
|
82
|
+
if (!sources.includes(name)) continue; // not compiled by anything; not required
|
|
83
|
+
required.add(rel(helper));
|
|
84
|
+
}
|
|
65
85
|
|
|
66
86
|
// Every runtime module. `files` ships "src" wholesale today; asserting each
|
|
67
87
|
// one catches anybody narrowing that later.
|
|
@@ -53,6 +53,15 @@ export const ALLOWED = [
|
|
|
53
53
|
/^com\.google\./i,
|
|
54
54
|
/^org\.swift\./i,
|
|
55
55
|
/^org\.json\./i,
|
|
56
|
+
// Build-tool namespaces, not app identifiers — and they arrive by the
|
|
57
|
+
// hundred the moment an Android project exists. `org.jetbrains.kotlin.android`
|
|
58
|
+
// is a Gradle plugin id, `org.jetbrains.kotlin:kotlin-gradle-plugin` a Maven
|
|
59
|
+
// coordinate, and `org.gradle.jvmargs` is a *property name* that merely looks
|
|
60
|
+
// like a bundle id. Added when the React Native testbed landed and flagged
|
|
61
|
+
// four of them; same judgement as the two lines above, which are Swift's and
|
|
62
|
+
// JSON's own namespaces rather than anybody's app.
|
|
63
|
+
/^org\.jetbrains\./i,
|
|
64
|
+
/^org\.gradle\./i,
|
|
56
65
|
/^com\.facebook\./i, // idb, a reference implementation named in the docs
|
|
57
66
|
/^com\.example\./i,
|
|
58
67
|
/^com\.acme\./i,
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
# The `integration` job, run here instead of there.
|
|
3
|
+
#
|
|
4
|
+
# One round trip on a hosted runner is 30+ minutes, and cancel-in-progress means
|
|
5
|
+
# a second push while you wait throws the answer away. Most of what that job
|
|
6
|
+
# asserts is reproducible on a developer's own simulator in a couple of minutes,
|
|
7
|
+
# so there is no reason to learn it from GitHub.
|
|
8
|
+
#
|
|
9
|
+
# What it cannot reproduce is the runner's *speed*: this machine settles a
|
|
10
|
+
# screen in a few hundred milliseconds where a loaded hosted runner has taken
|
|
11
|
+
# 45s for a two-step flow. So a green run here means "the code is right", not
|
|
12
|
+
# "CI will pass" — see DEFERRED 95. Everything else that has gone red, including
|
|
13
|
+
# the fingerprint collision that took an evening to find, would have shown up in
|
|
14
|
+
# this script.
|
|
15
|
+
#
|
|
16
|
+
# ./scripts/ci-integration-local.sh <udid>
|
|
17
|
+
#
|
|
18
|
+
# Assumes the device is booted and `simframe start` has been run on it, exactly
|
|
19
|
+
# as the job's earlier steps do.
|
|
20
|
+
set -u
|
|
21
|
+
DEVICE="${1:?usage: ci-integration-local.sh <udid>}"
|
|
22
|
+
cd "$(dirname "$0")/.." || exit 1
|
|
23
|
+
fails=0
|
|
24
|
+
step() { printf '\n=== %s ===\n' "$1"; }
|
|
25
|
+
ok() { printf 'ok %s\n' "$1"; }
|
|
26
|
+
bad() { printf 'FAIL %s\n' "$1"; fails=$((fails+1)); }
|
|
27
|
+
|
|
28
|
+
export SIMFRAME_STRICT=1
|
|
29
|
+
|
|
30
|
+
step "Every layer is the good one, or this fails"
|
|
31
|
+
node src/cli.js doctor --json --device="$DEVICE" > /tmp/doctor-local.json 2>/dev/null
|
|
32
|
+
node -e '
|
|
33
|
+
const d = JSON.parse(require("fs").readFileSync("/tmp/doctor-local.json", "utf8"));
|
|
34
|
+
const want = { "capture.engine": "simframed", "ocr.available": true };
|
|
35
|
+
let failed = false;
|
|
36
|
+
for (const [k, v] of Object.entries(want)) {
|
|
37
|
+
const good = d[k] === v;
|
|
38
|
+
console.log(`${good ? "ok " : "FAIL"} ${k} = ${JSON.stringify(d[k])} (want ${JSON.stringify(v)})`);
|
|
39
|
+
if (!good) failed = true;
|
|
40
|
+
}
|
|
41
|
+
for (const key of ["input.driver", "ax.driver"]) {
|
|
42
|
+
const good = d[key] === "simframed";
|
|
43
|
+
console.log(`${good ? "ok " : "FAIL"} ${key} = ${JSON.stringify(d[key])} (want "simframed")`);
|
|
44
|
+
if (!good) failed = true;
|
|
45
|
+
}
|
|
46
|
+
if (d.warnings > 0) {
|
|
47
|
+
console.log(`\n${d.warnings} degraded layer(s):`);
|
|
48
|
+
for (const c of d.checks.filter((c) => c.level !== "ok")) console.log(` ${c.level} ${c.name}: ${c.detail}`);
|
|
49
|
+
}
|
|
50
|
+
process.exit(failed ? 1 : 0);
|
|
51
|
+
' && ok "doctor: every layer is the daemon" || bad "doctor"
|
|
52
|
+
|
|
53
|
+
step "A step runs, and capture notices the screen change"
|
|
54
|
+
read_state() {
|
|
55
|
+
node src/cli.js state --device="$DEVICE" --json \
|
|
56
|
+
| node -e 'const s=JSON.parse(require("fs").readFileSync(0,"utf8")); process.stdout.write(s.hash+" "+s.seq)'
|
|
57
|
+
}
|
|
58
|
+
printf '[{"button":"home"},{"settle":true}]\n' > /tmp/reset-local.json
|
|
59
|
+
node src/cli.js do /tmp/reset-local.json --device="$DEVICE" >/dev/null 2>&1
|
|
60
|
+
BEFORE=$(read_state)
|
|
61
|
+
printf '[{"openUrl":"https://example.com"},{"settle":true}]\n' > /tmp/flow-local.json
|
|
62
|
+
if node src/cli.js do /tmp/flow-local.json --device="$DEVICE" >/dev/null 2>&1; then
|
|
63
|
+
AFTER=$(read_state)
|
|
64
|
+
if [ "${BEFORE%% *}" != "${AFTER%% *}" ]; then ok "frame hash changed: ${BEFORE%% *} -> ${AFTER%% *}"
|
|
65
|
+
else bad "a step ran cleanly but capture saw no change"; fi
|
|
66
|
+
else
|
|
67
|
+
bad "openUrl flow failed"
|
|
68
|
+
fi
|
|
69
|
+
|
|
70
|
+
step "The memory layer — screen map, refs, graph, verdicts, flows"
|
|
71
|
+
node scripts/ci-memory.mjs --device="$DEVICE" && ok "ci-memory" || bad "ci-memory"
|
|
72
|
+
|
|
73
|
+
step "The fingerprint distributions, measured and bounded"
|
|
74
|
+
node scripts/eval-fingerprint.mjs --tour=test/tours/device-native.json --rounds=3 \
|
|
75
|
+
--device="$DEVICE" --out=/tmp/fp-local-ci.json --label=local-ci >/tmp/fp-out.txt 2>&1 \
|
|
76
|
+
&& ok "fingerprint distributions" || { bad "fingerprint distributions"; tail -25 /tmp/fp-out.txt; }
|
|
77
|
+
|
|
78
|
+
printf '\n=== %s failure(s) ===\n' "$fails"
|
|
79
|
+
exit $((fails > 0))
|