simframe 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +137 -2
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +33 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1692 -44
- package/src/cli.js +131 -9
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +89 -7
- package/src/index.js +211 -14
- package/src/input.js +66 -3
- package/src/localhelper.js +155 -0
- package/src/matching.js +64 -1
- package/src/mcp.js +319 -27
- package/src/metrics.js +32 -3
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +2 -1
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +89 -9
- package/src/supervisor.js +117 -0
- package/src/view.js +335 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
// The local supervisor, and it may say exactly three things.
|
|
2
|
+
//
|
|
3
|
+
// Claude plans. The deterministic executor in src/actions.js runs the plan and
|
|
4
|
+
// verifies each step — round 6's best result was 18 steps in one call, and code
|
|
5
|
+
// is a better executor than a model: faster, exact, auditable, and it cannot
|
|
6
|
+
// hallucinate a step. What the executor has never had is judgement at the
|
|
7
|
+
// moment a step fails, so a batch died and a round trip was spent on a decision
|
|
8
|
+
// that was usually obvious.
|
|
9
|
+
//
|
|
10
|
+
// This answers that moment and nothing else:
|
|
11
|
+
//
|
|
12
|
+
// in {"goal":"...","step":"tap REVIEW","expected":"...","failure":"...","screen":["…labels…"],"stillMs":120}
|
|
13
|
+
// out {"decision":"wait","ms":410}
|
|
14
|
+
//
|
|
15
|
+
// `wait`, `retry`, `stop`. It cannot invent a step, skip one, substitute a
|
|
16
|
+
// target, or continue past an unexpected screen — not because a threshold
|
|
17
|
+
// forbids it but because those are not words it can say. That constraint is the
|
|
18
|
+
// safety property. `seek` was given latitude over *what* to open and pressed
|
|
19
|
+
// "YES, THIS FIXED MY PROBLEM" in a live app; a component whose whole answer
|
|
20
|
+
// space is three words cannot do that whatever it believes.
|
|
21
|
+
//
|
|
22
|
+
// Unavailable is a normal answer. It says so and exits, `doctor` reports
|
|
23
|
+
// `supervisor: none`, and the executor behaves exactly as it does today.
|
|
24
|
+
import Foundation
|
|
25
|
+
#if canImport(FoundationModels)
|
|
26
|
+
import FoundationModels
|
|
27
|
+
|
|
28
|
+
@available(macOS 26.0, *)
|
|
29
|
+
@Generable
|
|
30
|
+
enum Decision: String {
|
|
31
|
+
case wait
|
|
32
|
+
case retry
|
|
33
|
+
case stop
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
@available(macOS 26.0, *)
|
|
37
|
+
@Generable
|
|
38
|
+
struct Judgement {
|
|
39
|
+
@Guide(description: "wait if the screen is still arriving, retry if the same step should be attempted again, stop if nothing further can work")
|
|
40
|
+
var decision: Decision
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
struct Situation: Decodable {
|
|
44
|
+
let goal: String?
|
|
45
|
+
let step: String
|
|
46
|
+
let expected: String?
|
|
47
|
+
let failure: String
|
|
48
|
+
let screen: [String]?
|
|
49
|
+
let stillMs: Int?
|
|
50
|
+
let note: String?
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
func emit(_ object: [String: Any]) {
|
|
54
|
+
guard let data = try? JSONSerialization.data(withJSONObject: object),
|
|
55
|
+
let line = String(data: data, encoding: .utf8) else { return }
|
|
56
|
+
print(line)
|
|
57
|
+
fflush(stdout)
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
@available(macOS 26.0, *)
|
|
61
|
+
func serve() async {
|
|
62
|
+
switch SystemLanguageModel.default.availability {
|
|
63
|
+
case .available: break
|
|
64
|
+
case .unavailable(let reason):
|
|
65
|
+
emit(["unavailable": "\(reason)"])
|
|
66
|
+
return
|
|
67
|
+
@unknown default:
|
|
68
|
+
emit(["unavailable": "unknown availability"])
|
|
69
|
+
return
|
|
70
|
+
}
|
|
71
|
+
// Instructions, reused; the *session* is not.
|
|
72
|
+
//
|
|
73
|
+
// One session reused across requests is what made the supervisor go silent
|
|
74
|
+
// in the field: `LanguageModelSession` accumulates its transcript, so after
|
|
75
|
+
// seven real failures it hit `exceededContextWindowSize` — 4,441 tokens
|
|
76
|
+
// against a 4,096 maximum — and every request after that errored. The tester
|
|
77
|
+
// saw three rulings and then nothing for twenty supervised calls, while
|
|
78
|
+
// `doctor` in a separate process kept reporting the model available. It
|
|
79
|
+
// degraded before it broke, too: latency climbed 987ms to 1,890ms as the
|
|
80
|
+
// transcript grew, and the reasons collapsed into boilerplate repeated
|
|
81
|
+
// verbatim across unrelated failures.
|
|
82
|
+
//
|
|
83
|
+
// Each judgement is independent by nature — a failed step, what was
|
|
84
|
+
// expected, what is on screen — so there is nothing for a transcript to
|
|
85
|
+
// carry, and carrying it was pure cost even before it was fatal. The process
|
|
86
|
+
// stays warm, which is where the model load is paid; only the conversation
|
|
87
|
+
// is fresh.
|
|
88
|
+
let instructions = """
|
|
89
|
+
You supervise a UI test that is running a plan someone else wrote. A step \
|
|
90
|
+
has just failed. Decide one of three things and nothing else.
|
|
91
|
+
|
|
92
|
+
wait — the screen is still arriving: a spinner, a list that has not \
|
|
93
|
+
rendered, a count header with no rows, a transition in progress. Waiting \
|
|
94
|
+
would let the same step succeed.
|
|
95
|
+
retry — the step is sound and the moment was wrong: something moved under \
|
|
96
|
+
it, focus was lost, a stale reading. Repeating it would work.
|
|
97
|
+
stop — nothing further in the plan can work: the app is somewhere else, a \
|
|
98
|
+
required control is absent or disabled for a reason, or the screen has not \
|
|
99
|
+
responded at all.
|
|
100
|
+
|
|
101
|
+
Weigh the evidence you are given, in this order.
|
|
102
|
+
|
|
103
|
+
The plan's own guidance about this app comes FIRST and outranks everything \
|
|
104
|
+
below. It is the only knowledge of this app you have, and it was written by \
|
|
105
|
+
someone who has seen it. If it says lists arrive late, then a row that is \
|
|
106
|
+
not there yet means wait — even on a screen that has gone completely still, \
|
|
107
|
+
because a screen waiting on a network call is perfectly still and perfectly \
|
|
108
|
+
empty.
|
|
109
|
+
|
|
110
|
+
If a note says the screen is still filling in — a spinner, a count header \
|
|
111
|
+
with too few rows, a transition — that is decisive: answer wait.
|
|
112
|
+
If the screen has been still for under about 1500ms, content may still be \
|
|
113
|
+
arriving: prefer wait.
|
|
114
|
+
If the visible labels belong to a different part of the app than the step \
|
|
115
|
+
implies, answer stop.
|
|
116
|
+
If a control the step names is present but refused, answer stop.
|
|
117
|
+
Otherwise, if the step looks like it simply caught a bad moment, answer \
|
|
118
|
+
retry.
|
|
119
|
+
|
|
120
|
+
Never suggest a different step, a different target, or skipping ahead. You \
|
|
121
|
+
are not being asked what to do, only whether this can proceed. Answer with \
|
|
122
|
+
the decision alone.
|
|
123
|
+
"""
|
|
124
|
+
// A throwaway session up front so the first real judgement does not pay
|
|
125
|
+
// model load — measured at ~880ms against ~600ms warm.
|
|
126
|
+
_ = try? await LanguageModelSession(instructions: instructions)
|
|
127
|
+
.respond(to: "Step: warm\nIt failed with: warm", generating: Judgement.self)
|
|
128
|
+
emit(["ready": true])
|
|
129
|
+
while let line = readLine(strippingNewline: true) {
|
|
130
|
+
if line.isEmpty { continue }
|
|
131
|
+
guard let data = line.data(using: .utf8),
|
|
132
|
+
let s = try? JSONDecoder().decode(Situation.self, from: data) else {
|
|
133
|
+
emit(["error": "could not parse that line"])
|
|
134
|
+
continue
|
|
135
|
+
}
|
|
136
|
+
// Capped here as well as in the caller. A single realistic failure
|
|
137
|
+
// message carries the whole "Visible: …" list, and one request built
|
|
138
|
+
// from raw input measured 4,266 tokens against a 4,096 window — so a
|
|
139
|
+
// request can exceed the context on its own, with no transcript
|
|
140
|
+
// involved at all. Truncation belongs where the limit is.
|
|
141
|
+
let clip = { (t: String, n: Int) in t.count > n ? String(t.prefix(n)) + "…" : t }
|
|
142
|
+
var prompt = "Step: \(clip(s.step, 120))\nIt failed with: \(clip(s.failure, 220))"
|
|
143
|
+
if let goal = s.goal, !goal.isEmpty { prompt += "\nThe plan's guidance about this app: \(clip(goal, 300))" }
|
|
144
|
+
if let expected = s.expected, !expected.isEmpty { prompt += "\nExpected: \(clip(expected, 200))" }
|
|
145
|
+
if let ms = s.stillMs { prompt += "\nThe screen has been still for \(ms)ms" }
|
|
146
|
+
if let note = s.note, !note.isEmpty { prompt += "\nPerception note: \(clip(note, 160))" }
|
|
147
|
+
if let screen = s.screen, !screen.isEmpty {
|
|
148
|
+
prompt += "\nOn screen now: \(screen.prefix(14).map { clip($0, 28) }.joined(separator: ", "))"
|
|
149
|
+
}
|
|
150
|
+
let started = Date()
|
|
151
|
+
do {
|
|
152
|
+
let session = LanguageModelSession(instructions: instructions)
|
|
153
|
+
let out = try await session.respond(to: prompt, generating: Judgement.self)
|
|
154
|
+
// No reason field, deliberately. Asked for one it confabulated in
|
|
155
|
+
// every observed run: a correct `stop` justified as "screen is
|
|
156
|
+
// elsewhere" when the screen was exactly where the plan expected,
|
|
157
|
+
// and reasons repeated verbatim across unrelated failures. The
|
|
158
|
+
// reporter's judgement, and it is right — *"a right answer with a
|
|
159
|
+
// fabricated justification teaches me to distrust the
|
|
160
|
+
// justification, which is most of the value of it explaining
|
|
161
|
+
// itself. No reason at all would be better than a confident wrong
|
|
162
|
+
// one."* What the caller gets instead is which rule or which model
|
|
163
|
+
// answered, which is true by construction.
|
|
164
|
+
emit([
|
|
165
|
+
"decision": out.content.decision.rawValue,
|
|
166
|
+
"ms": Int(Date().timeIntervalSince(started) * 1000),
|
|
167
|
+
])
|
|
168
|
+
} catch {
|
|
169
|
+
emit(["error": "\(error)"])
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
if #available(macOS 26.0, *) {
|
|
175
|
+
await serve()
|
|
176
|
+
} else {
|
|
177
|
+
emit(["unavailable": "the on-device model needs macOS 26 or newer"])
|
|
178
|
+
}
|
|
179
|
+
#else
|
|
180
|
+
print("{\"unavailable\":\"this toolchain cannot import FoundationModels\"}")
|
|
181
|
+
#endif
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "simframe",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.11.0",
|
|
4
4
|
"mcpName": "io.github.lvlrSajjad/simframe",
|
|
5
5
|
"description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
|
|
6
6
|
"keywords": [
|
|
@@ -42,8 +42,11 @@
|
|
|
42
42
|
},
|
|
43
43
|
"files": [
|
|
44
44
|
"src",
|
|
45
|
+
"data",
|
|
45
46
|
"flows",
|
|
46
47
|
"native/ocr.swift",
|
|
48
|
+
"native/rank.swift",
|
|
49
|
+
"native/supervise.swift",
|
|
47
50
|
"native/simframed/Package.swift",
|
|
48
51
|
"native/simframed/Sources",
|
|
49
52
|
"native/simframed/Tests",
|
|
@@ -60,8 +60,28 @@ function requiredFiles() {
|
|
|
60
60
|
throw new Error('no Swift test sources found — has the layout moved? This check would silently pass.');
|
|
61
61
|
}
|
|
62
62
|
|
|
63
|
-
//
|
|
64
|
-
|
|
63
|
+
// Every standalone Swift helper compiled on demand from `src/`.
|
|
64
|
+
//
|
|
65
|
+
// This listed `native/ocr.swift` by name, which is why it did not catch the
|
|
66
|
+
// next two: `rank.swift` and `supervise.swift` were absent from `files`, so
|
|
67
|
+
// the local planner and the local supervisor would have been silently
|
|
68
|
+
// unavailable for every installed user — the exact failure this whole check
|
|
69
|
+
// exists to prevent, reproduced because the check named one file instead of
|
|
70
|
+
// stating the rule.
|
|
71
|
+
//
|
|
72
|
+
// The rule is: if a module under `src/` compiles it, it has to ship.
|
|
73
|
+
const helpers = walk(path.join(ROOT, 'native'), (p) => p.endsWith('.swift'))
|
|
74
|
+
.filter((p) => !p.includes(`${path.sep}simframed${path.sep}`) && !p.includes(`${path.sep}spikes${path.sep}`));
|
|
75
|
+
if (!helpers.length) {
|
|
76
|
+
throw new Error('no standalone Swift helper found under native/ — has the layout moved? This check would silently pass.');
|
|
77
|
+
}
|
|
78
|
+
const sources = walk(path.join(ROOT, 'src'), (p) => p.endsWith('.js'))
|
|
79
|
+
.map((f) => fs.readFileSync(f, 'utf8')).join('\n');
|
|
80
|
+
for (const helper of helpers) {
|
|
81
|
+
const name = path.basename(helper);
|
|
82
|
+
if (!sources.includes(name)) continue; // not compiled by anything; not required
|
|
83
|
+
required.add(rel(helper));
|
|
84
|
+
}
|
|
65
85
|
|
|
66
86
|
// Every runtime module. `files` ships "src" wholesale today; asserting each
|
|
67
87
|
// one catches anybody narrowing that later.
|
package/scripts/ci-memory.mjs
CHANGED
|
@@ -56,6 +56,36 @@ function check(ok, label, detail = '') {
|
|
|
56
56
|
return ok;
|
|
57
57
|
}
|
|
58
58
|
|
|
59
|
+
let skipped = 0;
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* A check whose *setup* did not happen, reported as untested rather than failed.
|
|
63
|
+
*
|
|
64
|
+
* This exists because the harness did to itself what three peer reports spent a
|
|
65
|
+
* day telling us not to do to callers. A `simctl launch` timed out on a loaded
|
|
66
|
+
* runner, `allowFail` swallowed it, and the next check announced `FAIL the
|
|
67
|
+
* screen actually changed before testing the stale ref — 7070f77757 ->
|
|
68
|
+
* 7070f77757`. Every word of that is true and it names the wrong thing: the
|
|
69
|
+
* screen did not change because **the app never launched**, which the run knew
|
|
70
|
+
* and did not say. Two more checks failed downstream of the same cause.
|
|
71
|
+
*
|
|
72
|
+
* A skip does not fail the build, and that is deliberate. A red build caused by
|
|
73
|
+
* somebody else's build farm is the cry-wolf failure this project keeps writing
|
|
74
|
+
* down: it trains everyone to re-run rather than to read. But it is counted and
|
|
75
|
+
* printed, because a run that tested less than it claims must say so.
|
|
76
|
+
*/
|
|
77
|
+
function skip(label, why) {
|
|
78
|
+
skipped += 1;
|
|
79
|
+
console.log(`skip ${label} — NOT TESTED: ${why}`);
|
|
80
|
+
return false;
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** Did a setup flow actually do what it was there for? */
|
|
84
|
+
function ran(res) {
|
|
85
|
+
if (!res || res.ok === false) return false;
|
|
86
|
+
return !(res.steps ?? []).some((st) => st.ok === false);
|
|
87
|
+
}
|
|
88
|
+
|
|
59
89
|
/**
|
|
60
90
|
* `expectFail` asserts a non-zero exit; `allowFail` tolerates one.
|
|
61
91
|
*
|
|
@@ -96,7 +126,21 @@ async function json(args, opts) {
|
|
|
96
126
|
* rather than serving a stale frame, which is the right behaviour and a
|
|
97
127
|
* transient condition. Retry those, and only those.
|
|
98
128
|
*/
|
|
129
|
+
// Capture dropping out, and — since 2026-09-11 — simctl timing out.
|
|
130
|
+
//
|
|
131
|
+
// The second one is this repo's oldest CI complaint and it was never in this
|
|
132
|
+
// pattern, so `jsonRetry` sailed past it: `simctl launch` takes 47-55s per
|
|
133
|
+
// attempt on a loaded hosted runner and `simctl openurl` times out internally,
|
|
134
|
+
// which means simframe is handed a failure it did not cause and cannot fix.
|
|
135
|
+
// Three checks in one run failed downstream of exactly that.
|
|
136
|
+
//
|
|
137
|
+
// Retried HERE and deliberately not inside simframe, which is the rule DEFERRED
|
|
138
|
+
// already wrote down for the bench script: a retried launch is an action that
|
|
139
|
+
// fires twice, and the verify barrier exists to stop simframe doing that on its
|
|
140
|
+
// own initiative. A test harness re-running its own setup is a different thing
|
|
141
|
+
// from a driver silently repeating a user's action.
|
|
99
142
|
const TRANSIENT = /did not produce a frame|display surface could not be read|no frames buffered/i;
|
|
143
|
+
const SIMCTL_FLAKE = /simctl|Command failed: xcrun|timed out/i;
|
|
100
144
|
|
|
101
145
|
async function jsonRetry(args, opts, attempts = 3) {
|
|
102
146
|
let last;
|
|
@@ -105,9 +149,13 @@ async function jsonRetry(args, opts, attempts = 3) {
|
|
|
105
149
|
return await json(args, opts);
|
|
106
150
|
} catch (err) {
|
|
107
151
|
last = err;
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
152
|
+
const capture = TRANSIENT.test(err.message);
|
|
153
|
+
const simctl = SIMCTL_FLAKE.test(err.message);
|
|
154
|
+
if (!capture && !simctl) throw err;
|
|
155
|
+
console.log(` (${capture ? 'capture dropped out' : 'simctl did not answer'}; retrying \`${args.join(' ')}\`)`);
|
|
156
|
+
// simctl's own timeouts are tens of seconds, so a 1.5s pause is not a
|
|
157
|
+
// wait, it is a formality. Give the runner room when that is the cause.
|
|
158
|
+
await new Promise((r) => setTimeout(r, simctl ? 5000 : 1500));
|
|
111
159
|
}
|
|
112
160
|
}
|
|
113
161
|
// Out of attempts on a capture error: the device is not blinking, it is gone.
|
|
@@ -242,13 +290,15 @@ if (first) {
|
|
|
242
290
|
// Now leave that screen WITHOUT re-reading it: `--json` skips the end-state
|
|
243
291
|
// map, so the ref table still describes the screen we have left.
|
|
244
292
|
const before = await markHash();
|
|
245
|
-
await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
|
|
293
|
+
const left = await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
|
|
246
294
|
const after = await markHash();
|
|
247
295
|
// Two different apps are two different screens by construction. The pixel
|
|
248
296
|
// hashes only have to agree with that, and they only get a say when they are
|
|
249
297
|
// informative enough to have one.
|
|
250
298
|
const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
|
|
251
|
-
|
|
299
|
+
if (!ran(left)) skip('the screen actually changed before testing the stale ref',
|
|
300
|
+
'the second app never launched, so there was no screen change to test against');
|
|
301
|
+
else check(moved, 'the screen actually changed before testing the stale ref',
|
|
252
302
|
`${before.slice(0, 10)} -> ${after.slice(0, 10)}`
|
|
253
303
|
+ (informativeHash(before) && informativeHash(after) ? '' : ' (degenerate hash: not evidence either way)'));
|
|
254
304
|
// Only assert the guard if the precondition actually held. Running it anyway
|
|
@@ -256,17 +306,28 @@ if (first) {
|
|
|
256
306
|
// screen, which is a false accusation against the one layer this file exists
|
|
257
307
|
// to defend — and it is how this check has failed twice.
|
|
258
308
|
if (moved) {
|
|
259
|
-
|
|
260
|
-
//
|
|
261
|
-
//
|
|
262
|
-
//
|
|
263
|
-
//
|
|
264
|
-
//
|
|
265
|
-
//
|
|
266
|
-
//
|
|
267
|
-
|
|
309
|
+
// This matched on prose twice and went red twice, both times for a refusal
|
|
310
|
+
// that was correct and better worded than the alternation knew — most
|
|
311
|
+
// recently `"Welcome to Reminders" is not on this screen`, which refuses
|
|
312
|
+
// *and* names what the number stood for. `find --json` now carries the
|
|
313
|
+
// reason as a field, so the check reads the contract instead of the
|
|
314
|
+
// sentence. What is under test is unchanged: the ref must not resolve to
|
|
315
|
+
// the coordinates it was numbered at on the screen we have left.
|
|
316
|
+
// A refusal that cannot be parsed is a failed check, not a dead script:
|
|
317
|
+
// `json` throws on anything non-JSON reaching the stream, and this is the
|
|
318
|
+
// one call site that expects a failure, so it is the one that would take
|
|
319
|
+
// the whole file down with it.
|
|
320
|
+
let stale;
|
|
321
|
+
try {
|
|
322
|
+
stale = await json(['find', `#${first.ref}`], { expectFail: true });
|
|
323
|
+
} catch (err) {
|
|
324
|
+
stale = { ok: null, error: err.message };
|
|
325
|
+
}
|
|
326
|
+
const refused = stale.ok === false
|
|
327
|
+
&& (stale.staleRef === true || stale.reason === 'unknown_screen' || stale.reason === 'ambiguous_intent');
|
|
328
|
+
check(refused,
|
|
268
329
|
'a ref numbered on another screen refuses instead of tapping those coordinates',
|
|
269
|
-
stale.
|
|
330
|
+
`${stale.reason ?? 'no reason'}${stale.staleRef ? ' staleRef' : ''} — ${String(stale.error ?? '').split('\n')[0].slice(0, 70)}`);
|
|
270
331
|
}
|
|
271
332
|
} else {
|
|
272
333
|
check(false, 'element refs', 'no elements to number');
|
|
@@ -343,7 +404,15 @@ const novelVerdicts = novelSteps.length
|
|
|
343
404
|
// accusing the layer underneath it. A device that crashed SpringBoard mid-run
|
|
344
405
|
// has not told us anything about the graph.
|
|
345
406
|
const novelRan = novelSteps.length > 0 && novelSteps.every((r) => r.ok !== false);
|
|
346
|
-
|
|
407
|
+
// The comment above states the principle and this line used to contradict it:
|
|
408
|
+
// it called `check`, so a `simctl openurl` that timed out on the runner failed
|
|
409
|
+
// the build and said "the novel action ran at all" as though the graph were at
|
|
410
|
+
// fault. It is a precondition. Untested is not broken.
|
|
411
|
+
if (!novelRan) {
|
|
412
|
+
skip('the novel action ran at all', `the action could not be dispatched — [${novelVerdicts.join(', ')}]`);
|
|
413
|
+
} else {
|
|
414
|
+
check(true, 'the novel action ran at all', `[${novelVerdicts.join(', ')}]`);
|
|
415
|
+
}
|
|
347
416
|
// The other half of the precondition, which was written above as a comment and
|
|
348
417
|
// then trusted. It is not trustworthy: the positioning run sends the device
|
|
349
418
|
// home, and a simulator that has been driven hard stops delivering `home` while
|
|
@@ -361,9 +430,20 @@ if (novelRan && novelMoved) {
|
|
|
361
430
|
'an action never taken here before is reported as unverified, not as verified',
|
|
362
431
|
`[${novelVerdicts.join(', ')}]`);
|
|
363
432
|
}
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
433
|
+
// The same precondition rule as the two above. The graph can only predict an
|
|
434
|
+
// outcome it has seen, and it can only have seen one if a pass actually ran —
|
|
435
|
+
// so a run in which every pass failed to dispatch says nothing about
|
|
436
|
+
// prediction. It failed the build as `pass 0` while the real cause was a
|
|
437
|
+
// simctl launch timing out, three checks upstream.
|
|
438
|
+
const anyPassRan = passes.some((p) => Array.isArray(p.run?.results) && p.run.results.some((r) => r.ok !== false));
|
|
439
|
+
if (!anyPassRan) {
|
|
440
|
+
skip('and once the graph has seen it, the outcome is predicted',
|
|
441
|
+
'no pass dispatched a step, so the graph was never given anything to learn');
|
|
442
|
+
} else {
|
|
443
|
+
check(passes.some((p) => p.verdicts.includes('ok')),
|
|
444
|
+
'and once the graph has seen it, the outcome is predicted',
|
|
445
|
+
`pass ${passes.findIndex((p) => p.verdicts.includes('ok')) + 1}`);
|
|
446
|
+
}
|
|
367
447
|
|
|
368
448
|
// One direction only: a wrong turn must fail the run. The converse does not
|
|
369
449
|
// hold — a run can fail for reasons that are not wrong turns, such as a step
|
|
@@ -472,5 +552,9 @@ try {
|
|
|
472
552
|
}
|
|
473
553
|
} catch { /* nothing to clean up */ }
|
|
474
554
|
|
|
475
|
-
|
|
555
|
+
const summary = [
|
|
556
|
+
failures ? `${failures} check(s) failed` : 'every check passed',
|
|
557
|
+
skipped ? `${skipped} check(s) NOT TESTED — the runner could not set them up` : null,
|
|
558
|
+
].filter(Boolean).join('; ');
|
|
559
|
+
console.log(`\n${summary}`);
|
|
476
560
|
process.exit(failures ? 1 : 0);
|
|
@@ -36,6 +36,7 @@ import { fileURLToPath } from 'node:url';
|
|
|
36
36
|
import * as fingerprint from '../src/fingerprint.js';
|
|
37
37
|
import * as matching from '../src/matching.js';
|
|
38
38
|
import * as analyze from '../src/analyze.js';
|
|
39
|
+
import * as view from '../src/view.js';
|
|
39
40
|
|
|
40
41
|
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
41
42
|
const DIR = path.join(ROOT, 'test', 'perception', 'screens');
|
|
@@ -68,6 +69,7 @@ export function checkScreen(fx) {
|
|
|
68
69
|
const authored = (expect.resolutions?.length ?? 0)
|
|
69
70
|
+ (expect.ambiguous?.length ?? 0)
|
|
70
71
|
+ (expect.none?.length ?? 0)
|
|
72
|
+
+ (expect.discoverable?.length ?? 0)
|
|
71
73
|
+ (fx.frame_pairs?.length ?? 0)
|
|
72
74
|
+ (expect.identity ? 1 : 0);
|
|
73
75
|
|
|
@@ -118,6 +120,37 @@ export function checkScreen(fx) {
|
|
|
118
120
|
}
|
|
119
121
|
}
|
|
120
122
|
|
|
123
|
+
// Discovery: does the default map still SAY what is on this screen?
|
|
124
|
+
//
|
|
125
|
+
// This check exists because of a regression it would have caught. A rule that
|
|
126
|
+
// dropped long non-interactive rows left every React Native list card out of
|
|
127
|
+
// the map — the cards expose their children as one concatenated accessibility
|
|
128
|
+
// label, so they look exactly like a Settings caption. Nothing became
|
|
129
|
+
// untappable, because `locate` reads targets rather than rows, so every
|
|
130
|
+
// resolution expectation above still passed. What was lost was the map's
|
|
131
|
+
// account of what is there, and an agent that cannot see a row falls back to
|
|
132
|
+
// a ~1600-token screenshot.
|
|
133
|
+
//
|
|
134
|
+
// Resolution and discovery are different claims, and the harness could only
|
|
135
|
+
// make the first one.
|
|
136
|
+
if (expect.discoverable?.length && screen) {
|
|
137
|
+
const { rows } = view.rowsFor({ targets }, { screen });
|
|
138
|
+
const shown = rows.map((r) => String(r.label ?? ''));
|
|
139
|
+
for (const want of expect.discoverable) {
|
|
140
|
+
// Compared as a prefix, because a long label is legitimately truncated in
|
|
141
|
+
// a row — truncated is discoverable, absent is not.
|
|
142
|
+
const head = want.slice(0, 24);
|
|
143
|
+
if (!shown.some((got) => got.startsWith(head))) {
|
|
144
|
+
findings.push({
|
|
145
|
+
kind: 'discovery',
|
|
146
|
+
query: `${want.slice(0, 40)}${want.length > 40 ? '…' : ''}`,
|
|
147
|
+
want: 'listed in the default map',
|
|
148
|
+
got: `${rows.length} row(s), none starting with it`,
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
121
154
|
// Identity drift. A token-rule change that silently alters a screen's
|
|
122
155
|
// structural hash discards every stored map and graph for it, and the failure
|
|
123
156
|
// is invisible — an old hash is a well-formed hash that matches nothing.
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Phase 17, go/no-go step 1: build the corpus and ask whether there is a job.
|
|
4
|
+
*
|
|
5
|
+
* The plan in docs/PHASES-HUMAN-PARITY.md said to export 200 `ambiguous_intent`
|
|
6
|
+
* and `no_plan` escalations. That corpus cannot be assembled: `no_plan` has
|
|
7
|
+
* never been logged once, `ambiguous_intent` is the reason the ranking bug
|
|
8
|
+
* produced (so the pre-fix records encode a bug), and the session id is minted
|
|
9
|
+
* per process — so a CLI-driven agent gets one "session" per command and the
|
|
10
|
+
* "filter to one session id" step has nothing to filter.
|
|
11
|
+
*
|
|
12
|
+
* There is a better corpus and it was there all along. Every verified graph edge
|
|
13
|
+
* is a decision that *worked*: the goal is in `step`, the screen it was taken on
|
|
14
|
+
* is the node, and the element list for that screen is in the screen map. That
|
|
15
|
+
* is (goal, element list, action) for every successful step, not just the rare
|
|
16
|
+
* failures — and the ground truth is stronger, because the tap was verified.
|
|
17
|
+
*
|
|
18
|
+
* What this measures is the question the whole phase turns on: of the decisions
|
|
19
|
+
* an agent actually makes, how many does the local matcher already resolve? A
|
|
20
|
+
* local planner can only earn its keep on the remainder. If the remainder is
|
|
21
|
+
* small, Phase 17 is a no-go regardless of how good the model is.
|
|
22
|
+
*
|
|
23
|
+
* Output is aggregate by design. This reads a real device's memory of real
|
|
24
|
+
* third-party apps, so it prints counts and never a label, an app name or a
|
|
25
|
+
* screen's contents. See the standing rule in scripts/check-private.mjs.
|
|
26
|
+
*
|
|
27
|
+
* Usage: node scripts/phase17-corpus.mjs [--device <udid>] [--json]
|
|
28
|
+
*/
|
|
29
|
+
import { readdirSync, readFileSync, existsSync } from 'node:fs';
|
|
30
|
+
import { join } from 'node:path';
|
|
31
|
+
import { homedir } from 'node:os';
|
|
32
|
+
import * as matching from '../src/matching.js';
|
|
33
|
+
import { goalOf } from '../src/actions.js';
|
|
34
|
+
|
|
35
|
+
const ROOT = process.env.SIMFRAME_HOME || join(homedir(), '.simframe');
|
|
36
|
+
|
|
37
|
+
const readJson = (p) => {
|
|
38
|
+
try { return JSON.parse(readFileSync(p, 'utf8')); } catch { return null; }
|
|
39
|
+
};
|
|
40
|
+
const listDir = (p) => (existsSync(p) ? readdirSync(p).filter((f) => f.endsWith('.json')) : []);
|
|
41
|
+
|
|
42
|
+
/** Screens are keyed by frame hash; graph nodes carry a layoutHash. Index both. */
|
|
43
|
+
function screenIndex(dev) {
|
|
44
|
+
const idx = new Map();
|
|
45
|
+
for (const f of listDir(join(dev, 'screens'))) {
|
|
46
|
+
const s = readJson(join(dev, 'screens', f));
|
|
47
|
+
if (!s?.targets?.length) continue;
|
|
48
|
+
for (const k of ['layoutHash', 'structuralHash', 'hash']) {
|
|
49
|
+
if (s[k] && !idx.has(s[k])) idx.set(s[k], s);
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
return idx;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Only some steps involve choosing an element. `launch` names an app, `button`
|
|
57
|
+
* names hardware, `openUrl` names a URL, `scroll` names a direction — none of
|
|
58
|
+
* them is a decision a planner could help with, and counting them was inflating
|
|
59
|
+
* the "not found" bucket to 45% on the first run of this script.
|
|
60
|
+
*/
|
|
61
|
+
const CHOOSES_AN_ELEMENT = new Set(['tap', 'type', 'scroll_to', 'scrollTo', 'assert', 'swipe_from', 'longPress']);
|
|
62
|
+
|
|
63
|
+
/** A "#3" is answered by the ref table and a "@x,y" by arithmetic. Neither is a decision. */
|
|
64
|
+
const isSelectorNotIntent = (goal) => /^#\d+$/.test(goal) || /^@?-?\d+\s*,\s*-?\d+$/.test(goal);
|
|
65
|
+
|
|
66
|
+
export function classify(dev) {
|
|
67
|
+
const idx = screenIndex(dev);
|
|
68
|
+
const out = {
|
|
69
|
+
screens_stored: idx.size,
|
|
70
|
+
edges: 0,
|
|
71
|
+
unjoinable: 0,
|
|
72
|
+
no_goal: 0,
|
|
73
|
+
not_an_element_step: 0,
|
|
74
|
+
selector_not_intent: 0,
|
|
75
|
+
cases: 0,
|
|
76
|
+
resolved: 0,
|
|
77
|
+
ambiguous: 0,
|
|
78
|
+
none: 0,
|
|
79
|
+
};
|
|
80
|
+
for (const f of listDir(join(dev, 'graph'))) {
|
|
81
|
+
const g = readJson(join(dev, 'graph', f));
|
|
82
|
+
const screen = idx.get(g?.layoutHash) ?? idx.get(g?.hash);
|
|
83
|
+
for (const e of g?.edges ?? []) {
|
|
84
|
+
out.edges += 1;
|
|
85
|
+
if (!screen) { out.unjoinable += 1; continue; }
|
|
86
|
+
const goal = goalOf(e.step);
|
|
87
|
+
if (!goal) { out.no_goal += 1; continue; }
|
|
88
|
+
if (!CHOOSES_AN_ELEMENT.has(e.step?.action)) { out.not_an_element_step += 1; continue; }
|
|
89
|
+
if (isSelectorNotIntent(String(goal).trim())) { out.selector_not_intent += 1; continue; }
|
|
90
|
+
out.cases += 1;
|
|
91
|
+
const r = matching.resolve(screen.targets, String(goal));
|
|
92
|
+
if (r.status === 'ok') out.resolved += 1;
|
|
93
|
+
else if (r.status === 'ambiguous') out.ambiguous += 1;
|
|
94
|
+
else out.none += 1;
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
return out;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* The second instrument, with the opposite bias.
|
|
102
|
+
*
|
|
103
|
+
* A verified edge is a decision that *succeeded*, so the graph cannot see the
|
|
104
|
+
* ones the matcher fumbled — those became escalations, Claude fixed them, and
|
|
105
|
+
* the edge was written with the corrected goal. Measuring the matcher on its
|
|
106
|
+
* own successes is partly circular. So also count resolution failures from the
|
|
107
|
+
* log against total edge traversals: same question, denominator built from
|
|
108
|
+
* failures instead of successes.
|
|
109
|
+
*/
|
|
110
|
+
export function resolutionFailureRate(dev) {
|
|
111
|
+
let traversals = 0;
|
|
112
|
+
for (const f of listDir(join(dev, 'graph'))) {
|
|
113
|
+
for (const e of readJson(join(dev, 'graph', f))?.edges ?? []) traversals += e.count ?? 0;
|
|
114
|
+
}
|
|
115
|
+
let failures = 0;
|
|
116
|
+
let escalations = 0;
|
|
117
|
+
const log = join(dev, 'escalations.jsonl');
|
|
118
|
+
if (existsSync(log)) {
|
|
119
|
+
for (const line of readFileSync(log, 'utf8').split('\n')) {
|
|
120
|
+
if (!line.trim()) continue;
|
|
121
|
+
let r;
|
|
122
|
+
try { r = JSON.parse(line); } catch { continue; }
|
|
123
|
+
escalations += 1;
|
|
124
|
+
if (r.reason === 'ambiguous_intent' || r.reason === 'unknown_screen') failures += 1;
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
return { traversals, escalations, failures };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const flags = process.argv.slice(2);
|
|
131
|
+
const only = flags.includes('--device') ? flags[flags.indexOf('--device') + 1] : null;
|
|
132
|
+
const devices = (existsSync(ROOT) ? readdirSync(ROOT) : [])
|
|
133
|
+
.filter((d) => !d.startsWith('TEST-') && d !== 'bin')
|
|
134
|
+
.filter((d) => existsSync(join(ROOT, d, 'graph')))
|
|
135
|
+
.filter((d) => !only || d === only);
|
|
136
|
+
|
|
137
|
+
const totals = { cases: 0, resolved: 0, ambiguous: 0, none: 0, edges: 0, unjoinable: 0, not_an_element_step: 0, selector_not_intent: 0 };
|
|
138
|
+
const rows = [];
|
|
139
|
+
for (const d of devices) {
|
|
140
|
+
const r = classify(join(ROOT, d));
|
|
141
|
+
if (!r.edges) continue;
|
|
142
|
+
rows.push({ device: `${d.slice(0, 8)}…`, ...r, ...resolutionFailureRate(join(ROOT, d)) });
|
|
143
|
+
for (const k of Object.keys(totals)) totals[k] += r[k];
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
if (flags.includes('--json')) {
|
|
147
|
+
console.log(JSON.stringify({ devices: rows, totals }, null, 2));
|
|
148
|
+
} else {
|
|
149
|
+
console.log('Phase 17 corpus — verified graph edges as (goal, element list, action)\n');
|
|
150
|
+
for (const r of rows) {
|
|
151
|
+
console.log(` ${r.device} edges ${r.edges} joinable ${r.edges - r.unjoinable} decisions ${r.cases}` +
|
|
152
|
+
` → matcher ok ${r.resolved}, ambiguous ${r.ambiguous}, not found ${r.none}`);
|
|
153
|
+
}
|
|
154
|
+
const { cases, resolved, ambiguous, none } = totals;
|
|
155
|
+
const pct = (n) => (cases ? `${Math.round((n / cases) * 1000) / 10}%` : '—');
|
|
156
|
+
console.log(`\n ${totals.edges} edges: ${totals.unjoinable} unjoinable, ` +
|
|
157
|
+
`${totals.not_an_element_step} chose no element (launch/button/openUrl/scroll), ` +
|
|
158
|
+
`${totals.selector_not_intent} used a #ref or a coordinate.`);
|
|
159
|
+
console.log(` TOTAL element decisions: ${cases}`);
|
|
160
|
+
console.log(` already resolved locally ${resolved} ${pct(resolved)} <- a planner adds nothing here`);
|
|
161
|
+
console.log(` ambiguous ${ambiguous} ${pct(ambiguous)} <- a planner could pick`);
|
|
162
|
+
console.log(` not found ${none} ${pct(none)} <- a planner cannot invent an element`);
|
|
163
|
+
const addressable = ambiguous;
|
|
164
|
+
console.log(`\n Addressable by a local planner: ${addressable} of ${cases} (${pct(addressable)}).`);
|
|
165
|
+
|
|
166
|
+
console.log('\nSecond instrument — resolution failures from the log, per traversal.');
|
|
167
|
+
console.log('A rate over 100% means the graph was discarded while the log kept appending');
|
|
168
|
+
console.log('(a MAP_VERSION or GRAPH_VERSION bump), so it is not a rate. Read the device');
|
|
169
|
+
console.log('that was driven by an agent doing real work, not the bench device.\n');
|
|
170
|
+
for (const r of rows) {
|
|
171
|
+
const rate = r.traversals ? `${Math.round((r.failures / r.traversals) * 1000) / 10}%` : '—';
|
|
172
|
+
console.log(` ${r.device} traversals ${String(r.traversals).padStart(4)}` +
|
|
173
|
+
` escalations ${String(r.escalations).padStart(4)}` +
|
|
174
|
+
` resolution failures ${String(r.failures).padStart(3)} ${rate.padStart(6)}`);
|
|
175
|
+
}
|
|
176
|
+
}
|