simframe 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +165 -5
- package/data/vocabulary/en.json +148 -0
- package/native/ocr.swift +13 -1
- package/native/rank.swift +87 -0
- package/native/simframed/Sources/PrivateAPI/CoreSimulatorPlatform.swift +43 -3
- package/native/simframed/Sources/PrivateAPI/PrivateAPI.swift +27 -0
- package/native/simframed/Sources/PrivateAPI/StubPlatform.swift +4 -0
- package/native/simframed/Sources/simframed/main.swift +13 -1
- package/native/supervise.swift +181 -0
- package/package.json +4 -1
- package/scripts/check-package.mjs +22 -2
- package/scripts/check-private.mjs +143 -0
- package/scripts/ci-memory.mjs +104 -20
- package/scripts/eval-perception.mjs +281 -0
- package/scripts/phase17-corpus.mjs +176 -0
- package/skills/simframe/SKILL.md +237 -5
- package/src/actions.js +1825 -38
- package/src/analyze.js +70 -0
- package/src/cli.js +214 -15
- package/src/control.js +1 -0
- package/src/fingerprint.js +19 -1
- package/src/graph.js +193 -11
- package/src/index.js +428 -16
- package/src/input.js +115 -8
- package/src/localhelper.js +155 -0
- package/src/matching.js +119 -3
- package/src/mcp.js +319 -27
- package/src/metrics.js +134 -8
- package/src/navigate.js +10 -7
- package/src/ocr.js +18 -1
- package/src/planner.js +195 -0
- package/src/platform/android.js +3 -2
- package/src/platform/ios.js +2 -1
- package/src/png.js +26 -0
- package/src/refs.js +51 -8
- package/src/regions.js +110 -1
- package/src/screenmap.js +109 -10
- package/src/supervisor.js +117 -0
- package/src/view.js +396 -7
- package/src/vocabulary.js +134 -0
- package/src/wrote.js +136 -0
package/README.md
CHANGED
|
@@ -96,8 +96,14 @@ ok accessibility tree simframed: AXPTranslator, host-side
|
|
|
96
96
|
ok on-device OCR available
|
|
97
97
|
ok booted simulator iPhone 17 Pro (iOS 26.5)
|
|
98
98
|
ok capture frame #888 322x700 in 2ms (age 538ms)
|
|
99
|
+
ok sensor mode full — accessibility and OCR fused on every read (~164ms)
|
|
100
|
+
ok local supervisor none — not requested (SIMFRAME_SUPERVISOR is unset)
|
|
101
|
+
ok local planner none — not requested (SIMFRAME_PLANNER is unset)
|
|
99
102
|
```
|
|
100
103
|
|
|
104
|
+
The last three are experiments and `none` is their normal answer. See
|
|
105
|
+
[Local tiers, off by default](#local-tiers-off-by-default).
|
|
106
|
+
|
|
101
107
|
### Claude Code
|
|
102
108
|
|
|
103
109
|
```bash
|
|
@@ -117,6 +123,75 @@ registered only for the directory you ran the command in.
|
|
|
117
123
|
}
|
|
118
124
|
```
|
|
119
125
|
|
|
126
|
+
## Recovering without a round trip
|
|
127
|
+
|
|
128
|
+
The measured cost of driving an app is not perception — warm, an
|
|
129
|
+
accessibility-only read is 85 ms and a fused read 142 ms. It is **round trips**:
|
|
130
|
+
in one instrumented run, 75% of the wall time was the agent thinking and the
|
|
131
|
+
call boundary, not simframe working. So the tools that matter most are the ones
|
|
132
|
+
that let a batch survive a problem instead of handing it back.
|
|
133
|
+
|
|
134
|
+
**Fallback selectors.** `{"tap": "Save", "or": ["Done", "Confirm"]}` — tried
|
|
135
|
+
locally in order, only an exhausted list reaching the model. Eligible after a
|
|
136
|
+
selector that did not *resolve* and nothing else, because retrying from a screen
|
|
137
|
+
you did not expect to be on is a second guess. A destructive-looking label is
|
|
138
|
+
refused as a substitute even if you list it.
|
|
139
|
+
|
|
140
|
+
**`{"seek": "change username", "budget": 6}`** opens containers, checks, and
|
|
141
|
+
comes back, depth first, inside a hard budget. It **acts** — opening a door
|
|
142
|
+
changes state — and it refuses to open anything that commits, abandons or
|
|
143
|
+
answers. It does not tap the target; it leaves you on the screen where the target
|
|
144
|
+
resolves.
|
|
145
|
+
|
|
146
|
+
**`{"sweep": "all", "fill": {…}}`** reads a long screen a viewport at a time and
|
|
147
|
+
fills each field while it is on screen. A form taller than the screen is only
|
|
148
|
+
knowable in pieces — the tree publishes what is rendered — and one scroll gesture
|
|
149
|
+
travels a non-deterministic distance, so finding a field and scrolling back to it
|
|
150
|
+
does not work. Sweeping does: on a real web form it filled every field in one
|
|
151
|
+
call. It detects both ends by measuring how far the *content* moved, ignoring
|
|
152
|
+
fixed chrome, which is the only reliable signal available since nothing reports a
|
|
153
|
+
scroll offset.
|
|
154
|
+
|
|
155
|
+
**`worked here before:`** puts the graph's own vocabulary in the map, most-used
|
|
156
|
+
first, rather than reporting a count. When the remembered controls are *not* on
|
|
157
|
+
the screen it says so instead, because that means two screens share one
|
|
158
|
+
fingerprint — and confident advice on a misidentified screen is how a remembered
|
|
159
|
+
label ends up pointing at a submit button.
|
|
160
|
+
|
|
161
|
+
## Local tiers, off by default
|
|
162
|
+
|
|
163
|
+
Two on-device model experiments, both `none` unless asked for, both degrading to
|
|
164
|
+
the existing matcher-then-model ladder, and CI runs with both off. They ship no
|
|
165
|
+
weights: Apple's Foundation Models framework has nothing to download, which is
|
|
166
|
+
the whole reason it clears this project's non-goal on shipping model weights.
|
|
167
|
+
|
|
168
|
+
| flag | what it does |
|
|
169
|
+
|---|---|
|
|
170
|
+
| `--sensor=ax-first` | read the accessibility tree alone (~50 ms) and pay for OCR only when a resolve fails |
|
|
171
|
+
| `--planner=apple` | order the containers `seek` opens; it cannot choose an action |
|
|
172
|
+
| `--supervisor=apple` | when a step fails, answer `wait`, `retry` or `stop` — nothing else — before the failure reaches the model |
|
|
173
|
+
|
|
174
|
+
All three are also per-call arguments on every MCP tool, because an MCP server's
|
|
175
|
+
environment is fixed when it spawns and comparing two modes inside one session
|
|
176
|
+
was otherwise impossible.
|
|
177
|
+
|
|
178
|
+
**What is measured and what is not.** The ranker: 5 of 6 top-1 on hand-written
|
|
179
|
+
cases, median 564 ms warm, and on a real exploration it went to the right region
|
|
180
|
+
in two steps where reading order wandered into version strings. The supervisor:
|
|
181
|
+
correct on four real batch-killers once the plan briefed it, 689–751 ms warm —
|
|
182
|
+
**on a bench, not in the field.** `ax-first` made no measurable difference to how
|
|
183
|
+
an agent drove a real app, with one small regression and one small win. Numbers
|
|
184
|
+
and conditions are in [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md); the judgements,
|
|
185
|
+
including a phase cancelled by its own measurement, are in
|
|
186
|
+
[`docs/DECISIONS.md`](docs/DECISIONS.md).
|
|
187
|
+
|
|
188
|
+
The supervisor's whole vocabulary is three words on purpose. It cannot invent a
|
|
189
|
+
step, skip one, substitute a target or continue past an unexpected screen — not
|
|
190
|
+
because a threshold forbids it but because those are not answers it can give.
|
|
191
|
+
That constraint replaced an earlier version of the same idea that was given
|
|
192
|
+
latitude over *what* to open and pressed a button labelled "YES, THIS FIXED MY
|
|
193
|
+
PROBLEM" in a live app.
|
|
194
|
+
|
|
120
195
|
## Capabilities are independent
|
|
121
196
|
|
|
122
197
|
Each layer works without the ones above it, and `doctor` tells you which you
|
|
@@ -255,10 +330,11 @@ nav-bar:
|
|
|
255
330
|
#2 text 201,64 Inbox
|
|
256
331
|
content:
|
|
257
332
|
#3 cell 201,140 Weekly digest
|
|
258
|
-
#4
|
|
333
|
+
#4 field 201,196 Search = weekly ~ weekly|
|
|
334
|
+
#5 switch 201,252 Notifications = 1
|
|
259
335
|
tab-bar:
|
|
260
|
-
#
|
|
261
|
-
#
|
|
336
|
+
#6 text 62,835 Inbox
|
|
337
|
+
#7 text 201,835 Settings
|
|
262
338
|
```
|
|
263
339
|
|
|
264
340
|
Region first, because "Inbox" the title and "Inbox" the tab differ only by where
|
|
@@ -267,14 +343,31 @@ is a selector: whatever this calls `#3`, the next call can tap as `#3` without
|
|
|
267
343
|
describing it. A ref is valid only while that screen is showing — used on a
|
|
268
344
|
different screen it refuses rather than tapping whatever now sits there.
|
|
269
345
|
|
|
346
|
+
`= something` is what the control *contains*, from the accessibility tree, and
|
|
347
|
+
`~ something` is what OCR read off the pixels. Both are printed, and where they
|
|
348
|
+
disagree that is the point: one is authoritative and the other is what is
|
|
349
|
+
actually on screen, and a field mid-edit can legitimately differ. A row with no
|
|
350
|
+
`=` is a control that reports no value, not an empty one.
|
|
351
|
+
|
|
352
|
+
When the elements were recalled from screen memory rather than looked at just
|
|
353
|
+
now, the header says so and how long ago — `elements recalled from 41s ago —
|
|
354
|
+
pass refresh for what is there now`. Identity is cached on purpose, because a
|
|
355
|
+
list with new rows is the same screen; contents are exactly what changes without
|
|
356
|
+
the screen changing, so the age is worth seeing.
|
|
357
|
+
|
|
270
358
|
Three ways to name a control, anywhere one is named:
|
|
271
359
|
|
|
272
360
|
| | |
|
|
273
361
|
| --- | --- |
|
|
274
|
-
|
|
|
275
|
-
| `
|
|
362
|
+
| `"Save"` · `the Assets tab` · `back` | **start here** — resolved by intent: verbs, typos, synonyms, icon-only controls by their common name |
|
|
363
|
+
| `#3` | the number the map gave it. Cheap and exact, but only within the round trip that numbered it |
|
|
276
364
|
| `@120,400` | raw point coordinates. Last resort: it cannot tell you it missed. |
|
|
277
365
|
|
|
366
|
+
The order is deliberate and it used to be the other way round. Four peer rounds
|
|
367
|
+
reported that intent resolution worked every time while refs renumbered
|
|
368
|
+
underneath them, so a table that led with `#3` and called it "unambiguous" was
|
|
369
|
+
recommending the more brittle of the two.
|
|
370
|
+
|
|
278
371
|
## Baselines: the thing to understand
|
|
279
372
|
|
|
280
373
|
Every change question is really "changed **since when**?" — and the answer is
|
|
@@ -542,8 +635,10 @@ simframe frame --out=now.png # newest frame, native resolution, to a file
|
|
|
542
635
|
simframe strip --count=6 # contact sheet, for an animation
|
|
543
636
|
simframe doctor --strict # any degraded layer is a non-zero exit
|
|
544
637
|
simframe escalations # why simframe still needs a model, by reason
|
|
638
|
+
simframe escalations --session # ...this agent only, not every agent on the device
|
|
545
639
|
simframe hpi # speed and accuracy against a human baseline
|
|
546
640
|
simframe baseline record settings-larger-text --runs=5 # record the human
|
|
641
|
+
simframe input reset # rebuild the HID session, without restarting anything
|
|
547
642
|
simframe start / status / stop [--force] / devices
|
|
548
643
|
simframe ui --device=emulator-5554 # or export SIMFRAME_DEVICE once
|
|
549
644
|
```
|
|
@@ -553,6 +648,46 @@ first: the space form set the flag to `true` and then resolved a device named
|
|
|
553
648
|
"true", which is a poor answer to a flag `doctor`'s own advice tells you to
|
|
554
649
|
type.
|
|
555
650
|
|
|
651
|
+
### Keeping a session cheap
|
|
652
|
+
|
|
653
|
+
The expensive part of driving a simulator with an agent is not the tapping, it
|
|
654
|
+
is the thinking between taps — observe, think, tap, observe, think. Measured
|
|
655
|
+
over one real session against a third-party app: **62 tool calls for 179 steps**,
|
|
656
|
+
and 48 of those calls were three steps or fewer. A twelve-step flow arrived as
|
|
657
|
+
five calls, and every boundary between them was a think.
|
|
658
|
+
|
|
659
|
+
Three things move that number, and simframe does the first two for you:
|
|
660
|
+
|
|
661
|
+
- **Batch.** `sim_do` runs a whole flow in one call, with an assert after each
|
|
662
|
+
step that matters. The asserts are what make it safe not to look in between:
|
|
663
|
+
a step that lands somewhere unplanned halts the flow instead of letting the
|
|
664
|
+
next four run against the wrong screen.
|
|
665
|
+
- **A `next:` line on every action result** — from the CLI and the MCP server
|
|
666
|
+
alike — computed locally from what the daemon already knows — whether the screen settled, whether the graph
|
|
667
|
+
recognises it, how many elements it has, whether any labels repeat. When it
|
|
668
|
+
says *nothing ambiguous — chain the next steps in one sim_do without looking
|
|
669
|
+
again*, that is the tool telling the agent it does not need to think.
|
|
670
|
+
- **A trailing map that was re-read, not recalled.** An action pays one
|
|
671
|
+
perception pass — a few hundred milliseconds, locally — so the map it returns
|
|
672
|
+
is the screen as it is now. The alternative was an agent spending a whole turn
|
|
673
|
+
on `ui --refresh` because it could not trust the one it was given.
|
|
674
|
+
- **One goal per session.** Sessions get slower with every turn. A flow that
|
|
675
|
+
runs as one call adds one exchange to the context instead of twelve.
|
|
676
|
+
|
|
677
|
+
And one thing to know about waiting: `settle` asks whether the screen stopped
|
|
678
|
+
moving, and a screen waiting on a network call has stopped moving. For anything
|
|
679
|
+
that arrives over the network, assert on the content you expect —
|
|
680
|
+
`{"waitFor": {"value": "Kate Bell"}}` — rather than on stillness. The map says
|
|
681
|
+
`STILL LOADING` when the classifier can see a load in flight, but only you know
|
|
682
|
+
what "arrived" means.
|
|
683
|
+
|
|
684
|
+
And the expensive habit worth naming: in that session, **28 of 62 calls returned
|
|
685
|
+
a screenshot** — about a third of its entire token cost — because the text map
|
|
686
|
+
could not report what a text field contained. It can now, so check the map
|
|
687
|
+
before reaching for pixels: a row carries the element's contents (`= Fryer 3`)
|
|
688
|
+
and its state (`disabled`), and the flow's own verdict already said whether the
|
|
689
|
+
action worked.
|
|
690
|
+
|
|
556
691
|
### The Claude Code skill
|
|
557
692
|
|
|
558
693
|
[`skills/simframe/SKILL.md`](skills/simframe/SKILL.md) teaches the CLI path
|
|
@@ -646,6 +781,31 @@ said a word — the exact failure shape, found by the thing built to catch it.
|
|
|
646
781
|
dramatically between visits will simply be rebuilt.
|
|
647
782
|
- It speeds up *confirming* a fix, not *locating* one. A bug living in a memo
|
|
648
783
|
comparator or a stale closure is not visible in any frame.
|
|
784
|
+
- A switch is tapped at the centre of its frame, and a switch's frame is the
|
|
785
|
+
whole row — so the tap lands on the label and the control, which sits at the
|
|
786
|
+
trailing end, does not move. Use `@x,y` on the control for now. Filed with
|
|
787
|
+
the measurement in `docs/DEFERRED.md`; it is a role-specific tap point, not a
|
|
788
|
+
patch at one call site.
|
|
789
|
+
- The simulator's display pipeline stops rendering under rapid app relaunch —
|
|
790
|
+
about six cycles, reproducibly — and every frame comes back black while
|
|
791
|
+
`simctl` itself reports success. simframe now says so instead of reading a
|
|
792
|
+
black screen as a calm one, but it cannot fix it: restarting the device is
|
|
793
|
+
the cure that always works, and it usually recovers on its own.
|
|
794
|
+
|
|
795
|
+
## What was decided, and what was not built
|
|
796
|
+
|
|
797
|
+
[`docs/DECISIONS.md`](docs/DECISIONS.md) is the register of judgements that
|
|
798
|
+
changed the plan: a phase cancelled by its own measurement, two features built
|
|
799
|
+
and reverted for cause, and the premises that turned out to be false. It is
|
|
800
|
+
short on purpose — the numbers live in
|
|
801
|
+
[`docs/BENCHMARKS.md`](docs/BENCHMARKS.md) and the open work in
|
|
802
|
+
[`docs/DEFERRED.md`](docs/DEFERRED.md).
|
|
803
|
+
|
|
804
|
+
The most useful entry is a **no-go**: Phase 17 proposed a small on-device model
|
|
805
|
+
to choose the next element, and measuring the prize before the model showed the
|
|
806
|
+
matcher already resolves 37 of 40 real decisions. By the time a step reaches
|
|
807
|
+
simframe the decision has already been made — the goal names the option,
|
|
808
|
+
because the agent chose it and then asked for it by name.
|
|
649
809
|
|
|
650
810
|
## Roadmap
|
|
651
811
|
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
{
|
|
2
|
+
"locale": "en",
|
|
3
|
+
"note": "Data, not code. CLAUDE.md: reflexes and the destructive vocabulary live in a locale-keyed file so adding a language is not a code change. Referenced by the verify barrier, per-step alternatives, exploration and (when built) reflexes.",
|
|
4
|
+
"destructive": {
|
|
5
|
+
"note": "Matched against a control's label, case-insensitively, on word boundaries. Nothing local may act on one of these: no retry, no alternative, no exploration, no reflex, no speculation. A model call is cheaper than a wrong tap.",
|
|
6
|
+
"words": [
|
|
7
|
+
"delete",
|
|
8
|
+
"remove",
|
|
9
|
+
"erase",
|
|
10
|
+
"clear",
|
|
11
|
+
"wipe",
|
|
12
|
+
"destroy",
|
|
13
|
+
"discard",
|
|
14
|
+
"pay",
|
|
15
|
+
"purchase",
|
|
16
|
+
"buy",
|
|
17
|
+
"subscribe",
|
|
18
|
+
"checkout",
|
|
19
|
+
"confirm payment",
|
|
20
|
+
"send",
|
|
21
|
+
"submit",
|
|
22
|
+
"post",
|
|
23
|
+
"publish",
|
|
24
|
+
"share",
|
|
25
|
+
"invite",
|
|
26
|
+
"sign out",
|
|
27
|
+
"log out",
|
|
28
|
+
"logout",
|
|
29
|
+
"deactivate",
|
|
30
|
+
"deregister",
|
|
31
|
+
"reset",
|
|
32
|
+
"restore",
|
|
33
|
+
"factory",
|
|
34
|
+
"unpair",
|
|
35
|
+
"forget",
|
|
36
|
+
"block",
|
|
37
|
+
"report",
|
|
38
|
+
"ban",
|
|
39
|
+
"unfriend",
|
|
40
|
+
"unfollow",
|
|
41
|
+
"cancel subscription",
|
|
42
|
+
"close account",
|
|
43
|
+
"delete account",
|
|
44
|
+
"accept",
|
|
45
|
+
"agree",
|
|
46
|
+
"allow",
|
|
47
|
+
"grant",
|
|
48
|
+
"trust",
|
|
49
|
+
"call",
|
|
50
|
+
"dial",
|
|
51
|
+
"transfer",
|
|
52
|
+
"withdraw",
|
|
53
|
+
"deposit",
|
|
54
|
+
"place order",
|
|
55
|
+
"confirm order",
|
|
56
|
+
"submit order",
|
|
57
|
+
"cancel order"
|
|
58
|
+
],
|
|
59
|
+
"notWords": {
|
|
60
|
+
"note": "Labels that are NOT destructive despite containing a destructive word. Matched against the WHOLE label, not as a phrase inside it: a bare 'Cancel' declines a dialog and must stay tappable, while 'Cancel order' is a different act and falls through to the list above. The bare noun 'order' is deliberately absent from that list for the same reason - 'Work Orders' and 'Order Details' are navigation, and only the verb phrase places one.",
|
|
61
|
+
"words": [
|
|
62
|
+
"cancel",
|
|
63
|
+
"clear all filters",
|
|
64
|
+
"clear search",
|
|
65
|
+
"clear filter"
|
|
66
|
+
]
|
|
67
|
+
}
|
|
68
|
+
},
|
|
69
|
+
"leavesTheApp": {
|
|
70
|
+
"note": "Nothing local may follow one of these either: it takes the run somewhere simframe was not asked to drive.",
|
|
71
|
+
"words": [
|
|
72
|
+
"open in safari",
|
|
73
|
+
"view in browser",
|
|
74
|
+
"app store",
|
|
75
|
+
"settings app",
|
|
76
|
+
"privacy policy",
|
|
77
|
+
"terms of service",
|
|
78
|
+
"contact support"
|
|
79
|
+
]
|
|
80
|
+
},
|
|
81
|
+
"exploration": {
|
|
82
|
+
"note": "Exploration is a different question from substitution, and conflating them cost a real run. `seek` opened CANCEL first — because 'cancel' is listed as safe above, so that a local tier may DECLINE a dialog — then AI TROUBLESHOOTING, then 'YES, THIS FIXED MY PROBLEM', ending five screens deep in a support chat with a half-built service request destroyed. May-I-tap-this-to-decline and may-I-open-this-as-a-door are not the same permission. This list answers the second, and it errs toward refusing: a door missed costs one step, a door taken wrongly costs the run.",
|
|
83
|
+
"neverOpen": [
|
|
84
|
+
"cancel",
|
|
85
|
+
"discard",
|
|
86
|
+
"abandon",
|
|
87
|
+
"close",
|
|
88
|
+
"dismiss",
|
|
89
|
+
"back",
|
|
90
|
+
"done",
|
|
91
|
+
"finish",
|
|
92
|
+
"yes",
|
|
93
|
+
"no",
|
|
94
|
+
"ok",
|
|
95
|
+
"okay",
|
|
96
|
+
"confirm",
|
|
97
|
+
"continue",
|
|
98
|
+
"proceed",
|
|
99
|
+
"next",
|
|
100
|
+
"skip",
|
|
101
|
+
"save",
|
|
102
|
+
"apply",
|
|
103
|
+
"submit",
|
|
104
|
+
"send",
|
|
105
|
+
"post",
|
|
106
|
+
"publish",
|
|
107
|
+
"accept",
|
|
108
|
+
"agree",
|
|
109
|
+
"allow",
|
|
110
|
+
"deny",
|
|
111
|
+
"block",
|
|
112
|
+
"grant",
|
|
113
|
+
"retry",
|
|
114
|
+
"try again",
|
|
115
|
+
"start over",
|
|
116
|
+
"help",
|
|
117
|
+
"support",
|
|
118
|
+
"contact",
|
|
119
|
+
"chat",
|
|
120
|
+
"feedback",
|
|
121
|
+
"rate",
|
|
122
|
+
"review us",
|
|
123
|
+
"sign in",
|
|
124
|
+
"sign up",
|
|
125
|
+
"log in",
|
|
126
|
+
"register",
|
|
127
|
+
"upgrade",
|
|
128
|
+
"buy",
|
|
129
|
+
"pay",
|
|
130
|
+
"activate",
|
|
131
|
+
"enable",
|
|
132
|
+
"disable",
|
|
133
|
+
"turn on",
|
|
134
|
+
"turn off",
|
|
135
|
+
"reset"
|
|
136
|
+
],
|
|
137
|
+
"neverOpenPatterns": [
|
|
138
|
+
"^yes[,. ]",
|
|
139
|
+
"^no[,. ]",
|
|
140
|
+
"^ok[,. ]",
|
|
141
|
+
"^don't ",
|
|
142
|
+
"^do not ",
|
|
143
|
+
"activate to ",
|
|
144
|
+
"^tap to ",
|
|
145
|
+
"^click to "
|
|
146
|
+
]
|
|
147
|
+
}
|
|
148
|
+
}
|
package/native/ocr.swift
CHANGED
|
@@ -7,8 +7,20 @@ guard args.count > 1, let img = NSImage(contentsOfFile: args[1]),
|
|
|
7
7
|
let cg = img.cgImage(forProposedRect: nil, context: nil, hints: nil) else {
|
|
8
8
|
FileHandle.standardError.write("cannot read image\n".data(using: .utf8)!); exit(1)
|
|
9
9
|
}
|
|
10
|
+
// Recognition level is a switch now, not a constant.
|
|
11
|
+
//
|
|
12
|
+
// CLAUDE.md fixes `.accurate` with language correction off, and that was chosen
|
|
13
|
+
// without a comparison — which is a threshold nobody had scored. `SIMFRAME_OCR`
|
|
14
|
+
// selects it so the two can be measured against each other on the perception
|
|
15
|
+
// harness: accuracy lost against milliseconds gained.
|
|
16
|
+
//
|
|
17
|
+
// The theory being tested is a human one: we read imprecisely and gain speed by
|
|
18
|
+
// it, tolerated by a forgiving match. simframe's matcher already forgives a
|
|
19
|
+
// great deal — prefixes, synonyms, typo distance, a Cyrillic-for-Latin fold —
|
|
20
|
+
// so a worse reading may cost nothing that matters.
|
|
10
21
|
let req = VNRecognizeTextRequest()
|
|
11
|
-
|
|
22
|
+
let level = ProcessInfo.processInfo.environment["SIMFRAME_OCR"]?.lowercased() ?? "accurate"
|
|
23
|
+
req.recognitionLevel = (level == "fast") ? .fast : .accurate
|
|
12
24
|
req.usesLanguageCorrection = false
|
|
13
25
|
try! VNImageRequestHandler(cgImage: cg, options: [:]).perform([req])
|
|
14
26
|
let w = Double(cg.width), h = Double(cg.height)
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
// A local ranker, and nothing more than a ranker.
|
|
2
|
+
//
|
|
3
|
+
// Reads one JSON object per line on stdin and writes one per line on stdout, so
|
|
4
|
+
// the caller pays model load once instead of once per question: measured, the
|
|
5
|
+
// first answer in a process costs ~880ms and every later one ~564ms.
|
|
6
|
+
//
|
|
7
|
+
// in {"goal":"change my username","options":["General","Account"]}
|
|
8
|
+
// out {"order":["Account","General"],"ms":564}
|
|
9
|
+
//
|
|
10
|
+
// It may only reorder the labels it was given. It never invents one, never says
|
|
11
|
+
// what to do, and never sees pixels — the caller has already decided that every
|
|
12
|
+
// option is permitted (see src/vocabulary.js) and will try them in some order
|
|
13
|
+
// regardless. This only decides which order.
|
|
14
|
+
//
|
|
15
|
+
// Unavailable is a normal answer, not an error: no Apple Intelligence, no Apple
|
|
16
|
+
// Silicon, an older macOS, or a user who has turned it off. It says so on the
|
|
17
|
+
// first line and exits, and `doctor` reports `planner: none` while the existing
|
|
18
|
+
// matcher-then-Claude ladder carries on unchanged.
|
|
19
|
+
import Foundation
|
|
20
|
+
#if canImport(FoundationModels)
|
|
21
|
+
import FoundationModels
|
|
22
|
+
|
|
23
|
+
@available(macOS 26.0, *)
|
|
24
|
+
@Generable
|
|
25
|
+
struct Ranking {
|
|
26
|
+
@Guide(description: "The given labels, most likely to lead to the goal first")
|
|
27
|
+
var order: [String]
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
struct Question: Decodable { let goal: String; let options: [String] }
|
|
31
|
+
|
|
32
|
+
func emit(_ object: [String: Any]) {
|
|
33
|
+
guard let data = try? JSONSerialization.data(withJSONObject: object),
|
|
34
|
+
let line = String(data: data, encoding: .utf8) else { return }
|
|
35
|
+
print(line)
|
|
36
|
+
fflush(stdout)
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
@available(macOS 26.0, *)
|
|
40
|
+
func serve() async {
|
|
41
|
+
switch SystemLanguageModel.default.availability {
|
|
42
|
+
case .available: break
|
|
43
|
+
case .unavailable(let reason):
|
|
44
|
+
emit(["unavailable": "\(reason)"])
|
|
45
|
+
return
|
|
46
|
+
@unknown default:
|
|
47
|
+
emit(["unavailable": "unknown availability"])
|
|
48
|
+
return
|
|
49
|
+
}
|
|
50
|
+
let session = LanguageModelSession(instructions: """
|
|
51
|
+
You rank user-interface controls. You are given something the user is looking \
|
|
52
|
+
for and a list of labels visible on one screen. Order the labels from most to \
|
|
53
|
+
least likely to lead to what they want. Use only the labels you were given, \
|
|
54
|
+
copied verbatim. Do not invent labels and do not explain.
|
|
55
|
+
""")
|
|
56
|
+
emit(["ready": true])
|
|
57
|
+
while let line = readLine(strippingNewline: true) {
|
|
58
|
+
if line.isEmpty { continue }
|
|
59
|
+
guard let data = line.data(using: .utf8),
|
|
60
|
+
let q = try? JSONDecoder().decode(Question.self, from: data) else {
|
|
61
|
+
emit(["error": "could not parse that line"])
|
|
62
|
+
continue
|
|
63
|
+
}
|
|
64
|
+
let started = Date()
|
|
65
|
+
do {
|
|
66
|
+
let answer = try await session.respond(
|
|
67
|
+
to: "Looking for: \(q.goal)\nLabels: \(q.options.joined(separator: ", "))",
|
|
68
|
+
generating: Ranking.self)
|
|
69
|
+
// Only labels we handed it, and never the same one twice: a model
|
|
70
|
+
// that paraphrases must not be able to smuggle in a new target.
|
|
71
|
+
var seen = Set<String>()
|
|
72
|
+
let kept = answer.content.order.filter { q.options.contains($0) && seen.insert($0).inserted }
|
|
73
|
+
emit(["order": kept, "ms": Int(Date().timeIntervalSince(started) * 1000)])
|
|
74
|
+
} catch {
|
|
75
|
+
emit(["error": "\(error)"])
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
if #available(macOS 26.0, *) {
|
|
81
|
+
await serve()
|
|
82
|
+
} else {
|
|
83
|
+
emit(["unavailable": "the on-device model needs macOS 26 or newer"])
|
|
84
|
+
}
|
|
85
|
+
#else
|
|
86
|
+
print("{\"unavailable\":\"this toolchain cannot import FoundationModels\"}")
|
|
87
|
+
#endif
|
|
@@ -307,11 +307,37 @@ public final class CoreSimulatorPlatform: SimulatorPlatform {
|
|
|
307
307
|
return raw is IOSurface
|
|
308
308
|
}
|
|
309
309
|
|
|
310
|
+
/// How long an on-demand surface read waits for a surface to exist.
|
|
311
|
+
///
|
|
312
|
+
/// The capture *loop* has re-resolve and rebind around this same call. This
|
|
313
|
+
/// path had nothing: it asked once and threw. On a loaded hosted runner
|
|
314
|
+
/// that cost a CI job — the display renders intermittently, one read landed
|
|
315
|
+
/// in a gap, text recognition reported "the display surface could not be
|
|
316
|
+
/// read", and the screen map came back empty. Four checks failed on a
|
|
317
|
+
/// device that was healthy before and after, and whose capture loop never
|
|
318
|
+
/// logged a single failure in the whole run.
|
|
319
|
+
static let surfaceRetryBudgetMs = 600
|
|
320
|
+
static let surfaceRetryStepMs = 50
|
|
321
|
+
|
|
310
322
|
public func withFrame<T>(_ body: (RawFrame) throws -> T) throws -> T {
|
|
311
323
|
guard let display else { throw PrivateAPIError.noDisplayPort }
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
324
|
+
var raw = display.perform(NSSelectorFromString("framebufferSurface"))?.takeUnretainedValue()
|
|
325
|
+
// Wait briefly rather than failing on the first miss. A display with
|
|
326
|
+
// nothing to draw can be between surfaces for a few tens of
|
|
327
|
+
// milliseconds, which is not the same condition as a display that has
|
|
328
|
+
// stopped rendering — and until now both said the same sentence.
|
|
329
|
+
var waitedMs = 0
|
|
330
|
+
while !(raw is IOSurface) && waitedMs < Self.surfaceRetryBudgetMs {
|
|
331
|
+
usleep(UInt32(Self.surfaceRetryStepMs) * 1000)
|
|
332
|
+
waitedMs += Self.surfaceRetryStepMs
|
|
333
|
+
raw = display.perform(NSSelectorFromString("framebufferSurface"))?.takeUnretainedValue()
|
|
334
|
+
}
|
|
335
|
+
guard let surface = raw as? IOSurface else {
|
|
336
|
+
// Two different conditions, and giving them one sentence bought two
|
|
337
|
+
// wrong diagnoses in a row: a transient miss reads exactly like the
|
|
338
|
+
// permanent wedge, so "it is the documented wedge, re-run it" was
|
|
339
|
+
// the advice both times. It said so here for the transient case.
|
|
340
|
+
throw PrivateAPIError.surfaceMissing(afterMs: waitedMs)
|
|
315
341
|
}
|
|
316
342
|
surface.lock(options: .readOnly, seed: nil)
|
|
317
343
|
defer { surface.unlock(options: .readOnly, seed: nil) }
|
|
@@ -573,6 +599,20 @@ extension CoreSimulatorPlatform {
|
|
|
573
599
|
guard hid != nil else { throw PrivateAPIError.hidUnavailable("could not rebuild the HID client") }
|
|
574
600
|
}
|
|
575
601
|
|
|
602
|
+
public func pressKey(usage: UInt32, modifiers: [UInt32]) throws {
|
|
603
|
+
let (hid, _) = try requireHID()
|
|
604
|
+
for m in modifiers { hid.key(usage: m, op: .down) }
|
|
605
|
+
defer { for m in modifiers.reversed() { hid.key(usage: m, op: .up) } }
|
|
606
|
+
// The usage-code path, not the character path. `type` sends characters
|
|
607
|
+
// and is therefore at the mercy of whichever keyboard layout iOS has
|
|
608
|
+
// active — which is why this device's own doctor warns about the fa and
|
|
609
|
+
// hy layouts. A usage code names a key *position* and is not translated,
|
|
610
|
+
// so Return is Return whatever is installed.
|
|
611
|
+
hid.key(usage: usage, op: .down)
|
|
612
|
+
Thread.sleep(forTimeInterval: 0.06)
|
|
613
|
+
hid.key(usage: usage, op: .up)
|
|
614
|
+
}
|
|
615
|
+
|
|
576
616
|
public func press(_ button: HardwareButton) throws {
|
|
577
617
|
let (hid, _) = try requireHID()
|
|
578
618
|
guard let code = HIDKeyboard.buttonCode(button) else {
|
|
@@ -52,6 +52,12 @@ public enum PrivateAPIError: Error, CustomStringConvertible {
|
|
|
52
52
|
case deviceNotFound(String)
|
|
53
53
|
case noDisplayPort
|
|
54
54
|
case surfaceUnavailable
|
|
55
|
+
/// The display had no surface *right now*, after waiting. Distinct from
|
|
56
|
+
/// `surfaceUnavailable`, which is the wedge: that one never heals without a
|
|
57
|
+
/// device restart, and this one is usually gone by the next frame. One
|
|
58
|
+
/// sentence for both is what made two sessions in a row diagnose a
|
|
59
|
+
/// transient miss as the wedge and advise a re-run.
|
|
60
|
+
case surfaceMissing(afterMs: Int)
|
|
55
61
|
case hidUnavailable(String)
|
|
56
62
|
case simctlFailed(String)
|
|
57
63
|
|
|
@@ -62,6 +68,10 @@ public enum PrivateAPIError: Error, CustomStringConvertible {
|
|
|
62
68
|
case .deviceNotFound(let u): return "no simulator matching \(u)"
|
|
63
69
|
case .noDisplayPort: return "the device exposes no active display port"
|
|
64
70
|
case .surfaceUnavailable: return "the display surface could not be read"
|
|
71
|
+
case .surfaceMissing(let ms):
|
|
72
|
+
return "no frame was available from the display for \(ms)ms"
|
|
73
|
+
+ " — this is usually transient; a display that has stopped rendering says"
|
|
74
|
+
+ " \"the display surface could not be read\" instead"
|
|
65
75
|
case .hidUnavailable(let d): return "input is unavailable: \(d)"
|
|
66
76
|
case .simctlFailed(let d): return "simctl \(d)"
|
|
67
77
|
}
|
|
@@ -122,6 +132,23 @@ public protocol SimulatorPlatform: AnyObject {
|
|
|
122
132
|
/// it is the only reliable route for content that must be exact.
|
|
123
133
|
func paste(_ text: String) throws
|
|
124
134
|
func press(_ button: HardwareButton) throws
|
|
135
|
+
/// Press one keyboard key by HID usage code.
|
|
136
|
+
///
|
|
137
|
+
/// Separate from `press`, which is the hardware buttons (HOME, LOCK, SIRI).
|
|
138
|
+
/// A peer was blocked outright for want of this: half of mobile search
|
|
139
|
+
/// fields submit on the keyboard return key, and there was no way to send
|
|
140
|
+
/// one. Typing "\n" as text goes through the active keyboard layout and
|
|
141
|
+
/// mangles the field instead — measured, it turned "Coke Display" into
|
|
142
|
+
/// "Coke In Display".
|
|
143
|
+
/// Press one keyboard key, optionally while holding modifiers.
|
|
144
|
+
///
|
|
145
|
+
/// Modifiers are usage codes too (Left GUI `0xE3`, Shift `0xE1`, Control
|
|
146
|
+
/// `0xE0`, Alt `0xE2`), held in order and released in reverse. This is what
|
|
147
|
+
/// makes clearing a field possible at all: nothing in XCUITest, Appium or
|
|
148
|
+
/// idb has a clear primitive, and the standard answer is Command-A followed
|
|
149
|
+
/// by Delete — which is layout-independent, because a modifier and Delete
|
|
150
|
+
/// are key *positions* and so is the `a` in Command-A.
|
|
151
|
+
func pressKey(usage: UInt32, modifiers: [UInt32]) throws
|
|
125
152
|
/// Rebuild the HID session.
|
|
126
153
|
///
|
|
127
154
|
/// Input has no feedback channel: a dispatched Indigo message reports
|
|
@@ -105,6 +105,10 @@ extension StubPlatform {
|
|
|
105
105
|
}
|
|
106
106
|
public func type(_ text: String) throws { recorded.append("type(\(text))") }
|
|
107
107
|
public func paste(_ text: String) throws { recorded.append("paste(\(text))") }
|
|
108
|
+
public func pressKey(usage: UInt32, modifiers: [UInt32]) throws {
|
|
109
|
+
throw PrivateAPIError.frameworksUnavailable("stub platform")
|
|
110
|
+
}
|
|
111
|
+
|
|
108
112
|
public func press(_ button: HardwareButton) throws { recorded.append("press(\(button.rawValue))") }
|
|
109
113
|
public func longPress(at point: CGPoint, durationMs: Double) throws {
|
|
110
114
|
recorded.append("longPress(\(Int(point.x)),\(Int(point.y)),\(Int(durationMs)))")
|
|
@@ -122,7 +122,7 @@ case "input":
|
|
|
122
122
|
print(" \(device.name): \(device.pixelWidth)x\(device.pixelHeight)px @\(device.scale)x = \(device.pointWidth)x\(device.pointHeight)pt")
|
|
123
123
|
} catch { fail("\(error)") }
|
|
124
124
|
|
|
125
|
-
case "tap", "swipe", "type", "paste", "press":
|
|
125
|
+
case "tap", "swipe", "type", "paste", "press", "key":
|
|
126
126
|
do {
|
|
127
127
|
_ = try platform.attach(udid: flag("udid"))
|
|
128
128
|
let positional = args.filter { !$0.hasPrefix("--") }.dropFirst()
|
|
@@ -141,6 +141,11 @@ case "tap", "swipe", "type", "paste", "press":
|
|
|
141
141
|
case "type":
|
|
142
142
|
guard !positional.isEmpty else { fail("usage: simframed type <text>") }
|
|
143
143
|
try platform.type(positional.joined(separator: " "))
|
|
144
|
+
case "key":
|
|
145
|
+
guard let u = positional.first, let usage = UInt32(u) else {
|
|
146
|
+
fail("usage: simframed key <hid-usage-code>")
|
|
147
|
+
}
|
|
148
|
+
try platform.pressKey(usage: usage, modifiers: [])
|
|
144
149
|
case "paste":
|
|
145
150
|
guard !positional.isEmpty else { fail("usage: simframed paste <text>") }
|
|
146
151
|
try platform.paste(positional.joined(separator: " "))
|
|
@@ -404,6 +409,13 @@ case "run":
|
|
|
404
409
|
}
|
|
405
410
|
try platform.press(button)
|
|
406
411
|
return done()
|
|
412
|
+
case "key":
|
|
413
|
+
guard let usage = request["usage"] as? NSNumber else {
|
|
414
|
+
return ["ok": false, "error": "key needs a HID usage code"]
|
|
415
|
+
}
|
|
416
|
+
let mods = (request["modifiers"] as? [NSNumber])?.map { $0.uint32Value } ?? []
|
|
417
|
+
try platform.pressKey(usage: usage.uint32Value, modifiers: mods)
|
|
418
|
+
return done()
|
|
407
419
|
case "resetInput":
|
|
408
420
|
try platform.resetInput()
|
|
409
421
|
FileHandle.standardError.write("simframed: HID session reset on request\n".data(using: .utf8)!)
|