simframe 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +165 -5
  2. package/data/vocabulary/en.json +148 -0
  3. package/native/ocr.swift +13 -1
  4. package/native/rank.swift +87 -0
  5. package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
  6. package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
  7. package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
  8. package/native/simframed/Sources/simframed/main.swift +13 -1
  9. package/native/supervise.swift +181 -0
  10. package/package.json +4 -1
  11. package/scripts/check-package.mjs +22 -2
  12. package/scripts/check-private.mjs +143 -0
  13. package/scripts/ci-memory.mjs +104 -20
  14. package/scripts/eval-perception.mjs +281 -0
  15. package/scripts/phase17-corpus.mjs +176 -0
  16. package/skills/simframe/SKILL.md +237 -5
  17. package/src/actions.js +1825 -38
  18. package/src/analyze.js +70 -0
  19. package/src/cli.js +214 -15
  20. package/src/control.js +1 -0
  21. package/src/fingerprint.js +19 -1
  22. package/src/graph.js +193 -11
  23. package/src/index.js +428 -16
  24. package/src/input.js +115 -8
  25. package/src/localhelper.js +155 -0
  26. package/src/matching.js +119 -3
  27. package/src/mcp.js +319 -27
  28. package/src/metrics.js +134 -8
  29. package/src/navigate.js +10 -7
  30. package/src/ocr.js +18 -1
  31. package/src/planner.js +195 -0
  32. package/src/platform/android.js +3 -2
  33. package/src/platform/ios.js +2 -1
  34. package/src/png.js +26 -0
  35. package/src/refs.js +51 -8
  36. package/src/regions.js +110 -1
  37. package/src/screenmap.js +109 -10
  38. package/src/supervisor.js +117 -0
  39. package/src/view.js +396 -7
  40. package/src/vocabulary.js +134 -0
  41. package/src/wrote.js +136 -0
@@ -0,0 +1,181 @@
1
+ // The local supervisor, and it may say exactly three things.
2
+ //
3
+ // Claude plans. The deterministic executor in src/actions.js runs the plan and
4
+ // verifies each step — round 6's best result was 18 steps in one call, and code
5
+ // is a better executor than a model: faster, exact, auditable, and it cannot
6
+ // hallucinate a step. What the executor has never had is judgement at the
7
+ // moment a step fails, so a batch died and a round trip was spent on a decision
8
+ // that was usually obvious.
9
+ //
10
+ // This answers that moment and nothing else:
11
+ //
12
+ // in {"goal":"...","step":"tap REVIEW","expected":"...","failure":"...","screen":["…labels…"],"stillMs":120}
13
+ // out {"decision":"wait","ms":410}
14
+ //
15
+ // `wait`, `retry`, `stop`. It cannot invent a step, skip one, substitute a
16
+ // target, or continue past an unexpected screen — not because a threshold
17
+ // forbids it but because those are not words it can say. That constraint is the
18
+ // safety property. `seek` was given latitude over *what* to open and pressed
19
+ // "YES, THIS FIXED MY PROBLEM" in a live app; a component whose whole answer
20
+ // space is three words cannot do that whatever it believes.
21
+ //
22
+ // Unavailable is a normal answer. It says so and exits, `doctor` reports
23
+ // `supervisor: none`, and the executor behaves exactly as it does today.
24
+ import Foundation
25
+ #if canImport(FoundationModels)
26
+ import FoundationModels
27
+
28
+ @available(macOS 26.0, *)
29
+ @Generable
30
+ enum Decision: String {
31
+ case wait
32
+ case retry
33
+ case stop
34
+ }
35
+
36
+ @available(macOS 26.0, *)
37
+ @Generable
38
+ struct Judgement {
39
+ @Guide(description: "wait if the screen is still arriving, retry if the same step should be attempted again, stop if nothing further can work")
40
+ var decision: Decision
41
+ }
42
+
43
+ struct Situation: Decodable {
44
+ let goal: String?
45
+ let step: String
46
+ let expected: String?
47
+ let failure: String
48
+ let screen: [String]?
49
+ let stillMs: Int?
50
+ let note: String?
51
+ }
52
+
53
+ func emit(_ object: [String: Any]) {
54
+ guard let data = try? JSONSerialization.data(withJSONObject: object),
55
+ let line = String(data: data, encoding: .utf8) else { return }
56
+ print(line)
57
+ fflush(stdout)
58
+ }
59
+
60
+ @available(macOS 26.0, *)
61
+ func serve() async {
62
+ switch SystemLanguageModel.default.availability {
63
+ case .available: break
64
+ case .unavailable(let reason):
65
+ emit(["unavailable": "\(reason)"])
66
+ return
67
+ @unknown default:
68
+ emit(["unavailable": "unknown availability"])
69
+ return
70
+ }
71
+ // Instructions, reused; the *session* is not.
72
+ //
73
+ // One session reused across requests is what made the supervisor go silent
74
+ // in the field: `LanguageModelSession` accumulates its transcript, so after
75
+ // seven real failures it hit `exceededContextWindowSize` — 4,441 tokens
76
+ // against a 4,096 maximum — and every request after that errored. The tester
77
+ // saw three rulings and then nothing for twenty supervised calls, while
78
+ // `doctor` in a separate process kept reporting the model available. It
79
+ // degraded before it broke, too: latency climbed 987ms to 1,890ms as the
80
+ // transcript grew, and the reasons collapsed into boilerplate repeated
81
+ // verbatim across unrelated failures.
82
+ //
83
+ // Each judgement is independent by nature — a failed step, what was
84
+ // expected, what is on screen — so there is nothing for a transcript to
85
+ // carry, and carrying it was pure cost even before it was fatal. The process
86
+ // stays warm, which is where the model load is paid; only the conversation
87
+ // is fresh.
88
+ let instructions = """
89
+ You supervise a UI test that is running a plan someone else wrote. A step \
90
+ has just failed. Decide one of three things and nothing else.
91
+
92
+ wait — the screen is still arriving: a spinner, a list that has not \
93
+ rendered, a count header with no rows, a transition in progress. Waiting \
94
+ would let the same step succeed.
95
+ retry — the step is sound and the moment was wrong: something moved under \
96
+ it, focus was lost, a stale reading. Repeating it would work.
97
+ stop — nothing further in the plan can work: the app is somewhere else, a \
98
+ required control is absent or disabled for a reason, or the screen has not \
99
+ responded at all.
100
+
101
+ Weigh the evidence you are given, in this order.
102
+
103
+ The plan's own guidance about this app comes FIRST and outranks everything \
104
+ below. It is the only knowledge of this app you have, and it was written by \
105
+ someone who has seen it. If it says lists arrive late, then a row that is \
106
+ not there yet means wait — even on a screen that has gone completely still, \
107
+ because a screen waiting on a network call is perfectly still and perfectly \
108
+ empty.
109
+
110
+ If a note says the screen is still filling in — a spinner, a count header \
111
+ with too few rows, a transition — that is decisive: answer wait.
112
+ If the screen has been still for under about 1500ms, content may still be \
113
+ arriving: prefer wait.
114
+ If the visible labels belong to a different part of the app than the step \
115
+ implies, answer stop.
116
+ If a control the step names is present but refused, answer stop.
117
+ Otherwise, if the step looks like it simply caught a bad moment, answer \
118
+ retry.
119
+
120
+ Never suggest a different step, a different target, or skipping ahead. You \
121
+ are not being asked what to do, only whether this can proceed. Answer with \
122
+ the decision alone.
123
+ """
124
+ // A throwaway session up front so the first real judgement does not pay
125
+ // model load — measured at ~880ms against ~600ms warm.
126
+ _ = try? await LanguageModelSession(instructions: instructions)
127
+ .respond(to: "Step: warm\nIt failed with: warm", generating: Judgement.self)
128
+ emit(["ready": true])
129
+ while let line = readLine(strippingNewline: true) {
130
+ if line.isEmpty { continue }
131
+ guard let data = line.data(using: .utf8),
132
+ let s = try? JSONDecoder().decode(Situation.self, from: data) else {
133
+ emit(["error": "could not parse that line"])
134
+ continue
135
+ }
136
+ // Capped here as well as in the caller. A single realistic failure
137
+ // message carries the whole "Visible: …" list, and one request built
138
+ // from raw input measured 4,266 tokens against a 4,096 window — so a
139
+ // request can exceed the context on its own, with no transcript
140
+ // involved at all. Truncation belongs where the limit is.
141
+ let clip = { (t: String, n: Int) in t.count > n ? String(t.prefix(n)) + "…" : t }
142
+ var prompt = "Step: \(clip(s.step, 120))\nIt failed with: \(clip(s.failure, 220))"
143
+ if let goal = s.goal, !goal.isEmpty { prompt += "\nThe plan's guidance about this app: \(clip(goal, 300))" }
144
+ if let expected = s.expected, !expected.isEmpty { prompt += "\nExpected: \(clip(expected, 200))" }
145
+ if let ms = s.stillMs { prompt += "\nThe screen has been still for \(ms)ms" }
146
+ if let note = s.note, !note.isEmpty { prompt += "\nPerception note: \(clip(note, 160))" }
147
+ if let screen = s.screen, !screen.isEmpty {
148
+ prompt += "\nOn screen now: \(screen.prefix(14).map { clip($0, 28) }.joined(separator: ", "))"
149
+ }
150
+ let started = Date()
151
+ do {
152
+ let session = LanguageModelSession(instructions: instructions)
153
+ let out = try await session.respond(to: prompt, generating: Judgement.self)
154
+ // No reason field, deliberately. Asked for one it confabulated in
155
+ // every observed run: a correct `stop` justified as "screen is
156
+ // elsewhere" when the screen was exactly where the plan expected,
157
+ // and reasons repeated verbatim across unrelated failures. The
158
+ // reporter's judgement, and it is right — *"a right answer with a
159
+ // fabricated justification teaches me to distrust the
160
+ // justification, which is most of the value of it explaining
161
+ // itself. No reason at all would be better than a confident wrong
162
+ // one."* What the caller gets instead is which rule or which model
163
+ // answered, which is true by construction.
164
+ emit([
165
+ "decision": out.content.decision.rawValue,
166
+ "ms": Int(Date().timeIntervalSince(started) * 1000),
167
+ ])
168
+ } catch {
169
+ emit(["error": "\(error)"])
170
+ }
171
+ }
172
+ }
173
+
174
+ if #available(macOS 26.0, *) {
175
+ await serve()
176
+ } else {
177
+ emit(["unavailable": "the on-device model needs macOS 26 or newer"])
178
+ }
179
+ #else
180
+ print("{\"unavailable\":\"this toolchain cannot import FoundationModels\"}")
181
+ #endif
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "simframe",
3
- "version": "0.9.0",
3
+ "version": "0.11.0",
4
4
  "mcpName": "io.github.lvlrSajjad/simframe",
5
5
  "description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
6
6
  "keywords": [
@@ -42,8 +42,11 @@
42
42
  },
43
43
  "files": [
44
44
  "src",
45
+ "data",
45
46
  "flows",
46
47
  "native/ocr.swift",
48
+ "native/rank.swift",
49
+ "native/supervise.swift",
47
50
  "native/simframed/Package.swift",
48
51
  "native/simframed/Sources",
49
52
  "native/simframed/Tests",
@@ -60,8 +60,28 @@ function requiredFiles() {
60
60
  throw new Error('no Swift test sources found — has the layout moved? This check would silently pass.');
61
61
  }
62
62
 
63
- // The standalone OCR helper, compiled on demand by src/ocr.js.
64
- required.add(rel(path.join(ROOT, 'native', 'ocr.swift')));
63
+ // Every standalone Swift helper compiled on demand from `src/`.
64
+ //
65
+ // This listed `native/ocr.swift` by name, which is why it did not catch the
66
+ // next two: `rank.swift` and `supervise.swift` were absent from `files`, so
67
+ // the local planner and the local supervisor would have been silently
68
+ // unavailable for every installed user — the exact failure this whole check
69
+ // exists to prevent, reproduced because the check named one file instead of
70
+ // stating the rule.
71
+ //
72
+ // The rule is: if a module under `src/` compiles it, it has to ship.
73
+ const helpers = walk(path.join(ROOT, 'native'), (p) => p.endsWith('.swift'))
74
+ .filter((p) => !p.includes(`${path.sep}simframed${path.sep}`) && !p.includes(`${path.sep}spikes${path.sep}`));
75
+ if (!helpers.length) {
76
+ throw new Error('no standalone Swift helper found under native/ — has the layout moved? This check would silently pass.');
77
+ }
78
+ const sources = walk(path.join(ROOT, 'src'), (p) => p.endsWith('.js'))
79
+ .map((f) => fs.readFileSync(f, 'utf8')).join('\n');
80
+ for (const helper of helpers) {
81
+ const name = path.basename(helper);
82
+ if (!sources.includes(name)) continue; // not compiled by anything; not required
83
+ required.add(rel(helper));
84
+ }
65
85
 
66
86
  // Every runtime module. `files` ships "src" wholesale today; asserting each
67
87
  // one catches anybody narrowing that later.
@@ -0,0 +1,143 @@
1
+ #!/usr/bin/env node
2
+ // No third-party app identifiers in this repository. Ever, from anyone.
3
+ //
4
+ // simframe is a general-purpose tool: you install it and Claude Code drives
5
+ // *your* app on the simulator. It has no relationship with any particular app,
6
+ // so no particular app's bundle id belongs in it — not in the source, not in
7
+ // the docs, and not in a secret either. A denylist of specific strings would
8
+ // assume there is one app to protect, which is the wrong shape for this.
9
+ //
10
+ // So the rule is a pattern, not a list, and it needs no configuration at all.
11
+ // Anything shaped like a reverse-DNS bundle id is flagged unless it is one of:
12
+ //
13
+ // * a platform's own com.apple.*, com.android.*, com.google.*
14
+ // * a documentation placeholder com.example.*, com.acme.*, com.mycompany.*
15
+ // * this project's own identifiers
16
+ //
17
+ // That works on a fresh clone, on a fork, and in a pull request from a stranger,
18
+ // which a secret does not. An optional `.private-strings` file (gitignored) or
19
+ // $SIMFRAME_PRIVATE_STRINGS still adds extra patterns for anyone who wants them,
20
+ // but nothing depends on either existing.
21
+ //
22
+ // How this got written: a 282-line field-notes file about a real third-party app
23
+ // was committed here by `git add -A` and pushed, an hour after the first version
24
+ // of this script was written to prevent exactly that. It could not fire, because
25
+ // it was waiting for a denylist nobody had supplied. A guard with a
26
+ // precondition is a guard that is off.
27
+ //
28
+ // Nothing here ever prints a match. It prints the file and the line number, so
29
+ // the output of a failed run is safe to paste into an issue, a CI log, or a
30
+ // conversation with an agent — which is where the last one would have gone.
31
+ import fs from 'node:fs';
32
+ import path from 'node:path';
33
+ import { execFileSync } from 'node:child_process';
34
+ import { fileURLToPath } from 'node:url';
35
+
36
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
37
+ const LIST_FILE = path.join(ROOT, '.private-strings');
38
+
39
+ /**
40
+ * Bundle-id-shaped strings, and the ones that are fine.
41
+ *
42
+ * The first segment is restricted to real reverse-DNS prefixes, which is what
43
+ * keeps ordinary property chains out: `res.state.seq`, `registry.paths.dir` and
44
+ * `import.meta.url` all look exactly like bundle ids until you require the head
45
+ * to be a TLD.
46
+ */
47
+ const BUNDLE = /\b(?:com|io|org|net|dev|co|app|me|xyz|uk|de|fr|jp|nl|se|ca|au)\.[A-Za-z][A-Za-z0-9_-]{1,30}(?:\.[A-Za-z][A-Za-z0-9_-]{0,30}){1,3}\b/g;
48
+
49
+ /** Platform-owned, placeholder, or ours. Anything else is somebody's app. */
50
+ export const ALLOWED = [
51
+ /^com\.apple\./i,
52
+ /^com\.android\./i,
53
+ /^com\.google\./i,
54
+ /^org\.swift\./i,
55
+ /^org\.json\./i,
56
+ /^com\.facebook\./i, // idb, a reference implementation named in the docs
57
+ /^com\.example\./i,
58
+ /^com\.acme\./i,
59
+ /^com\.mycompany\./i,
60
+ /^com\.yourcompany\./i,
61
+ /^io\.github\./i,
62
+ ];
63
+
64
+ export function isAllowedIdentifier(id) {
65
+ return ALLOWED.some((re) => re.test(id));
66
+ }
67
+
68
+ /** Extra patterns, for anyone who wants them. Nothing depends on this existing. */
69
+ export function patternsFrom({ env, file } = {}) {
70
+ const raw = [
71
+ ...String(env ?? '').split(/[\n,]/),
72
+ ...String(file ?? '').split(/\n/),
73
+ ];
74
+ return [...new Set(
75
+ raw
76
+ .map((s) => s.trim())
77
+ .filter((s) => s && !s.startsWith('#'))
78
+ // One character would match everything, which is a check that only ever
79
+ // fails and therefore only ever gets disabled.
80
+ .filter((s) => s.length >= 3)
81
+ .map((s) => s.toLowerCase()),
82
+ )];
83
+ }
84
+
85
+ /** Which lines of `text` are a problem. Line numbers only, never matches. */
86
+ export function offendingLines(text, patterns = []) {
87
+ const hits = [];
88
+ const lines = String(text).split('\n');
89
+ for (let i = 0; i < lines.length; i += 1) {
90
+ const lower = lines[i].toLowerCase();
91
+ const why = [];
92
+ const extra = patterns.filter((p) => lower.includes(p)).length;
93
+ if (extra) why.push(`${extra} denied pattern(s)`);
94
+ const ids = (lines[i].match(BUNDLE) ?? []).filter((id) => !isAllowedIdentifier(id));
95
+ // Reported as a count and a shape, never as the identifier: knowing which
96
+ // app leaked is worth less to a bug report than not restating it.
97
+ if (ids.length) why.push(`${ids.length} third-party bundle id(s)`);
98
+ if (why.length) hits.push({ line: i + 1, why: why.join(', ') });
99
+ }
100
+ return hits;
101
+ }
102
+
103
+ function trackedFiles() {
104
+ return execFileSync('git', ['ls-files', '-z'], { cwd: ROOT, encoding: 'utf8' })
105
+ .split('\0')
106
+ .filter(Boolean);
107
+ }
108
+
109
+ function main() {
110
+ const patterns = patternsFrom({
111
+ env: process.env.SIMFRAME_PRIVATE_STRINGS,
112
+ file: fs.existsSync(LIST_FILE) ? fs.readFileSync(LIST_FILE, 'utf8') : '',
113
+ });
114
+ const files = trackedFiles();
115
+ let failed = 0;
116
+ for (const rel of files) {
117
+ const full = path.join(ROOT, rel);
118
+ let text;
119
+ try {
120
+ const stat = fs.statSync(full);
121
+ if (!stat.isFile() || stat.size > 4_000_000) continue;
122
+ text = fs.readFileSync(full, 'utf8');
123
+ } catch {
124
+ continue;
125
+ }
126
+ // A NUL byte means this is not text, and a substring hit in it is noise.
127
+ if (text.includes('\0')) continue;
128
+ for (const hit of offendingLines(text, patterns)) {
129
+ console.error(`check-private: ${rel}:${hit.line} — ${hit.why}`);
130
+ failed += 1;
131
+ }
132
+ }
133
+ console.log(`check-private: ${files.length} tracked file(s); no third-party bundle ids`
134
+ + (patterns.length ? `, plus ${patterns.length} local pattern(s)` : '')
135
+ + ` — ${failed} line(s) flagged`);
136
+ if (failed) {
137
+ console.error('check-private: nothing above prints the match itself. '
138
+ + 'A third-party app identifier does not belong in a general-purpose tool.');
139
+ process.exitCode = 1;
140
+ }
141
+ }
142
+
143
+ if (import.meta.url === `file://${process.argv[1]}`) main();
@@ -56,6 +56,36 @@ function check(ok, label, detail = '') {
56
56
  return ok;
57
57
  }
58
58
 
59
+ let skipped = 0;
60
+
61
+ /**
62
+ * A check whose *setup* did not happen, reported as untested rather than failed.
63
+ *
64
+ * This exists because the harness did to itself what three peer reports spent a
65
+ * day telling us not to do to callers. A `simctl launch` timed out on a loaded
66
+ * runner, `allowFail` swallowed it, and the next check announced `FAIL the
67
+ * screen actually changed before testing the stale ref — 7070f77757 ->
68
+ * 7070f77757`. Every word of that is true and it names the wrong thing: the
69
+ * screen did not change because **the app never launched**, which the run knew
70
+ * and did not say. Two more checks failed downstream of the same cause.
71
+ *
72
+ * A skip does not fail the build, and that is deliberate. A red build caused by
73
+ * somebody else's build farm is the cry-wolf failure this project keeps writing
74
+ * down: it trains everyone to re-run rather than to read. But it is counted and
75
+ * printed, because a run that tested less than it claims must say so.
76
+ */
77
+ function skip(label, why) {
78
+ skipped += 1;
79
+ console.log(`skip ${label} — NOT TESTED: ${why}`);
80
+ return false;
81
+ }
82
+
83
+ /** Did a setup flow actually do what it was there for? */
84
+ function ran(res) {
85
+ if (!res || res.ok === false) return false;
86
+ return !(res.steps ?? []).some((st) => st.ok === false);
87
+ }
88
+
59
89
  /**
60
90
  * `expectFail` asserts a non-zero exit; `allowFail` tolerates one.
61
91
  *
@@ -96,7 +126,21 @@ async function json(args, opts) {
96
126
  * rather than serving a stale frame, which is the right behaviour and a
97
127
  * transient condition. Retry those, and only those.
98
128
  */
129
+ // Capture dropping out, and — since 2026-09-11 — simctl timing out.
130
+ //
131
+ // The second one is this repo's oldest CI complaint and it was never in this
132
+ // pattern, so `jsonRetry` sailed past it: `simctl launch` takes 47-55s per
133
+ // attempt on a loaded hosted runner and `simctl openurl` times out internally,
134
+ // which means simframe is handed a failure it did not cause and cannot fix.
135
+ // Three checks in one run failed downstream of exactly that.
136
+ //
137
+ // Retried HERE and deliberately not inside simframe, which is the rule DEFERRED
138
+ // already wrote down for the bench script: a retried launch is an action that
139
+ // fires twice, and the verify barrier exists to stop simframe doing that on its
140
+ // own initiative. A test harness re-running its own setup is a different thing
141
+ // from a driver silently repeating a user's action.
99
142
  const TRANSIENT = /did not produce a frame|display surface could not be read|no frames buffered/i;
143
+ const SIMCTL_FLAKE = /simctl|Command failed: xcrun|timed out/i;
100
144
 
101
145
  async function jsonRetry(args, opts, attempts = 3) {
102
146
  let last;
@@ -105,9 +149,13 @@ async function jsonRetry(args, opts, attempts = 3) {
105
149
  return await json(args, opts);
106
150
  } catch (err) {
107
151
  last = err;
108
- if (!TRANSIENT.test(err.message)) throw err;
109
- console.log(` (capture dropped out; retrying \`${args.join(' ')}\`)`);
110
- await new Promise((r) => setTimeout(r, 1500));
152
+ const capture = TRANSIENT.test(err.message);
153
+ const simctl = SIMCTL_FLAKE.test(err.message);
154
+ if (!capture && !simctl) throw err;
155
+ console.log(` (${capture ? 'capture dropped out' : 'simctl did not answer'}; retrying \`${args.join(' ')}\`)`);
156
+ // simctl's own timeouts are tens of seconds, so a 1.5s pause is not a
157
+ // wait, it is a formality. Give the runner room when that is the cause.
158
+ await new Promise((r) => setTimeout(r, simctl ? 5000 : 1500));
111
159
  }
112
160
  }
113
161
  // Out of attempts on a capture error: the device is not blinking, it is gone.
@@ -242,13 +290,15 @@ if (first) {
242
290
  // Now leave that screen WITHOUT re-reading it: `--json` skips the end-state
243
291
  // map, so the ref table still describes the screen we have left.
244
292
  const before = await markHash();
245
- await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
293
+ const left = await jsonRetry(['do', AT_OTHER_SCREEN], { allowFail: true });
246
294
  const after = await markHash();
247
295
  // Two different apps are two different screens by construction. The pixel
248
296
  // hashes only have to agree with that, and they only get a say when they are
249
297
  // informative enough to have one.
250
298
  const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
251
- check(moved, 'the screen actually changed before testing the stale ref',
299
+ if (!ran(left)) skip('the screen actually changed before testing the stale ref',
300
+ 'the second app never launched, so there was no screen change to test against');
301
+ else check(moved, 'the screen actually changed before testing the stale ref',
252
302
  `${before.slice(0, 10)} -> ${after.slice(0, 10)}`
253
303
  + (informativeHash(before) && informativeHash(after) ? '' : ' (degenerate hash: not evidence either way)'));
254
304
  // Only assert the guard if the precondition actually held. Running it anyway
@@ -256,17 +306,28 @@ if (first) {
256
306
  // screen, which is a false accusation against the one layer this file exists
257
307
  // to defend — and it is how this check has failed twice.
258
308
  if (moved) {
259
- const stale = await cli(['find', `#${first.ref}`], { expectFail: true });
260
- // Matched on prose, which is this check's weakness: simframe refused
261
- // correctly with "#1 cannot be trusted here — simframe does not recognise
262
- // this screen. Read it again", and the check failed because that wording was
263
- // not one of the two it knew. The refusal is what matters, so the
264
- // alternation covers how a refusal is actually phrased; the durable fix is a
265
- // machine-readable reason on the failure, which `find --json` does not yet
266
- // carry.
267
- check(/different screen|read the screen|read it again|does not recognise this screen|cannot be trusted/i.test(stale),
309
+ // This matched on prose twice and went red twice, both times for a refusal
310
+ // that was correct and better worded than the alternation knew — most
311
+ // recently `"Welcome to Reminders" is not on this screen`, which refuses
312
+ // *and* names what the number stood for. `find --json` now carries the
313
+ // reason as a field, so the check reads the contract instead of the
314
+ // sentence. What is under test is unchanged: the ref must not resolve to
315
+ // the coordinates it was numbered at on the screen we have left.
316
+ // A refusal that cannot be parsed is a failed check, not a dead script:
317
+ // `json` throws on anything non-JSON reaching the stream, and this is the
318
+ // one call site that expects a failure, so it is the one that would take
319
+ // the whole file down with it.
320
+ let stale;
321
+ try {
322
+ stale = await json(['find', `#${first.ref}`], { expectFail: true });
323
+ } catch (err) {
324
+ stale = { ok: null, error: err.message };
325
+ }
326
+ const refused = stale.ok === false
327
+ && (stale.staleRef === true || stale.reason === 'unknown_screen' || stale.reason === 'ambiguous_intent');
328
+ check(refused,
268
329
  'a ref numbered on another screen refuses instead of tapping those coordinates',
269
- stale.trim().split('\n')[0]?.slice(0, 90));
330
+ `${stale.reason ?? 'no reason'}${stale.staleRef ? ' staleRef' : ''} — ${String(stale.error ?? '').split('\n')[0].slice(0, 70)}`);
270
331
  }
271
332
  } else {
272
333
  check(false, 'element refs', 'no elements to number');
@@ -343,7 +404,15 @@ const novelVerdicts = novelSteps.length
343
404
  // accusing the layer underneath it. A device that crashed SpringBoard mid-run
344
405
  // has not told us anything about the graph.
345
406
  const novelRan = novelSteps.length > 0 && novelSteps.every((r) => r.ok !== false);
346
- check(novelRan, 'the novel action ran at all', `[${novelVerdicts.join(', ')}]`);
407
+ // The comment above states the principle and this line used to contradict it:
408
+ // it called `check`, so a `simctl openurl` that timed out on the runner failed
409
+ // the build and said "the novel action ran at all" as though the graph were at
410
+ // fault. It is a precondition. Untested is not broken.
411
+ if (!novelRan) {
412
+ skip('the novel action ran at all', `the action could not be dispatched — [${novelVerdicts.join(', ')}]`);
413
+ } else {
414
+ check(true, 'the novel action ran at all', `[${novelVerdicts.join(', ')}]`);
415
+ }
347
416
  // The other half of the precondition, which was written above as a comment and
348
417
  // then trusted. It is not trustworthy: the positioning run sends the device
349
418
  // home, and a simulator that has been driven hard stops delivering `home` while
@@ -361,9 +430,20 @@ if (novelRan && novelMoved) {
361
430
  'an action never taken here before is reported as unverified, not as verified',
362
431
  `[${novelVerdicts.join(', ')}]`);
363
432
  }
364
- check(passes.some((p) => p.verdicts.includes('ok')),
365
- 'and once the graph has seen it, the outcome is predicted',
366
- `pass ${passes.findIndex((p) => p.verdicts.includes('ok')) + 1}`);
433
+ // The same precondition rule as the two above. The graph can only predict an
434
+ // outcome it has seen, and it can only have seen one if a pass actually ran —
435
+ // so a run in which every pass failed to dispatch says nothing about
436
+ // prediction. It failed the build as `pass 0` while the real cause was a
437
+ // simctl launch timing out, three checks upstream.
438
+ const anyPassRan = passes.some((p) => Array.isArray(p.run?.results) && p.run.results.some((r) => r.ok !== false));
439
+ if (!anyPassRan) {
440
+ skip('and once the graph has seen it, the outcome is predicted',
441
+ 'no pass dispatched a step, so the graph was never given anything to learn');
442
+ } else {
443
+ check(passes.some((p) => p.verdicts.includes('ok')),
444
+ 'and once the graph has seen it, the outcome is predicted',
445
+ `pass ${passes.findIndex((p) => p.verdicts.includes('ok')) + 1}`);
446
+ }
367
447
 
368
448
  // One direction only: a wrong turn must fail the run. The converse does not
369
449
  // hold — a run can fail for reasons that are not wrong turns, such as a step
@@ -472,5 +552,9 @@ try {
472
552
  }
473
553
  } catch { /* nothing to clean up */ }
474
554
 
475
- console.log(`\n${failures ? `${failures} check(s) failed` : 'every check passed'}`);
555
+ const summary = [
556
+ failures ? `${failures} check(s) failed` : 'every check passed',
557
+ skipped ? `${skipped} check(s) NOT TESTED — the runner could not set them up` : null,
558
+ ].filter(Boolean).join('; ');
559
+ console.log(`\n${summary}`);
476
560
  process.exit(failures ? 1 : 0);