simframe 0.12.0 → 0.12.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +40 -0
- package/package.json +1 -1
- package/scripts/ci-memory.mjs +6 -1
- package/scripts/collect-rulings.mjs +14 -0
- package/scripts/eval-fingerprint.mjs +28 -2
- package/scripts/replay-rulings.mjs +147 -0
- package/scripts/score-rulings.mjs +125 -0
- package/src/actions.js +62 -9
- package/src/cli.js +33 -5
- package/src/index.js +42 -3
- package/src/metrics.js +19 -1
- package/src/navigate.js +28 -2
- package/src/ollama.js +204 -0
- package/src/platform/ios.js +47 -2
- package/src/screenmap.js +76 -0
- package/src/supervisor.js +88 -3
package/README.md
CHANGED
|
@@ -185,6 +185,33 @@ and conditions are in [`docs/BENCHMARKS.md`](docs/BENCHMARKS.md); the judgements
|
|
|
185
185
|
including a phase cancelled by its own measurement, are in
|
|
186
186
|
[`docs/DECISIONS.md`](docs/DECISIONS.md).
|
|
187
187
|
|
|
188
|
+
**And the measurements that changed our minds** are in
|
|
189
|
+
[`docs/EXPERIMENTS.md`](docs/EXPERIMENTS.md), with what we expected beforehand
|
|
190
|
+
written down beside each — including the capacity comparison the supervisor
|
|
191
|
+
design was assumed to make unnecessary. Asked the same 22 situations, three
|
|
192
|
+
times each:
|
|
193
|
+
|
|
194
|
+
| arm | accuracy | median | deterministic |
|
|
195
|
+
|---|---|---|---|
|
|
196
|
+
| always the commonest answer | 55% | — | — |
|
|
197
|
+
| **`stillMs > 3000ms`, no model at all** | **95%** | **0 ms** | yes |
|
|
198
|
+
| Apple Foundation Models (~3B, on-device) | 77 / 82 / 86% | ~640 ms | **no** |
|
|
199
|
+
| `qwen3:8b` via Ollama (4-bit, 5.2 GB) | **91%** | 919 ms | yes |
|
|
200
|
+
| `qwen3:14b` via Ollama (4-bit, 9.3 GB) | 82% | 1,489 ms | yes |
|
|
201
|
+
|
|
202
|
+
Three results, and the third is the one that matters. The larger model scored
|
|
203
|
+
*lower* than the smaller one. A free threshold on a number the daemon already
|
|
204
|
+
computes beat all three, on a population it was never fitted to. And **the
|
|
205
|
+
accuracy ranking inverts the safety ranking**: every arm errs in one direction
|
|
206
|
+
only, Apple's errors are all `wait` where `stop` was right and both Qwen arms'
|
|
207
|
+
are all `stop` where `wait` was right — and a wrong `stop` abandons a plan that
|
|
208
|
+
would have worked, while a wrong `wait` costs a settle. A comparison reporting
|
|
209
|
+
only the percentages would have recommended the wrong model.
|
|
210
|
+
|
|
211
|
+
The Ollama arm is an **experiment, not a recommendation**: off unless named,
|
|
212
|
+
no weights shipped, no dependency added, and every arm reads the same briefing
|
|
213
|
+
out of `native/supervise.swift` so no arm is answering a different question.
|
|
214
|
+
|
|
188
215
|
The supervisor's whole vocabulary is three words on purpose. It cannot invent a
|
|
189
216
|
step, skip one, substitute a target or continue past an unexpected screen — not
|
|
190
217
|
because a threshold forbids it but because those are not answers it can give.
|
|
@@ -610,6 +637,19 @@ accepted as the name that loop used to have.
|
|
|
610
637
|
radio button moves ~0.1 % of the screen — below the change threshold — which
|
|
611
638
|
used to burn the full timeout. Now the step returns in ~3 s marked
|
|
612
639
|
`[no visible change]`, so you know to check rather than wait.
|
|
640
|
+
- **And a gesture aimed at a coordinate says what it landed on.** A tap or swipe
|
|
641
|
+
that changed nothing prints the element covering its start point, because a
|
|
642
|
+
gesture goes to whatever is on top there:
|
|
643
|
+
|
|
644
|
+
```
|
|
645
|
+
swiped 83,141 -> 83,800 [no visible change]
|
|
646
|
+
[the swipe start point 83,141 is inside "Settings" (nav-bar)]
|
|
647
|
+
```
|
|
648
|
+
|
|
649
|
+
Reported from the field as fifteen minutes lost to six identical swipes that
|
|
650
|
+
a support banner was swallowing — a banner that was *in the element list the
|
|
651
|
+
same call printed*. Nothing said "there is something at y≈753 and you started
|
|
652
|
+
at y=750".
|
|
613
653
|
|
|
614
654
|
## CLI
|
|
615
655
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "simframe",
|
|
3
|
-
"version": "0.12.
|
|
3
|
+
"version": "0.12.2",
|
|
4
4
|
"mcpName": "io.github.lvlrSajjad/simframe",
|
|
5
5
|
"description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
|
|
6
6
|
"keywords": [
|
package/scripts/ci-memory.mjs
CHANGED
|
@@ -491,7 +491,12 @@ const walked = await jsonRetry(['goto', target.hash], { allowFail: true });
|
|
|
491
491
|
// neither walking there nor naming why it cannot, so this check failed with an
|
|
492
492
|
// empty detail — the reason was `undefined` — and the check was right to fail.
|
|
493
493
|
// Now there is a name for it, and the list has to know the name.
|
|
494
|
-
|
|
494
|
+
// `route-halted` and `arrived-elsewhere` are new: a walk that ran and did not
|
|
495
|
+
// land used to return `{ok: false}` with no reason at all, which failed this
|
|
496
|
+
// very check with an empty detail. It was the one outcome here nobody had
|
|
497
|
+
// named, and the check found it.
|
|
498
|
+
const outcomes = ['no-route', 'unreplayable-edge', 'ambiguous', 'unknown-screen', 'no-identity',
|
|
499
|
+
'route-halted', 'arrived-elsewhere'];
|
|
495
500
|
check(walked.ok === true || outcomes.includes(walked.reason),
|
|
496
501
|
'and asked for a screen it knows, it either walks there or names why it cannot',
|
|
497
502
|
walked.ok ? (walked.already ? 'already there' : `walked ${walked.ranSteps} step(s)`) : walked.reason);
|
|
@@ -310,3 +310,17 @@ if (unattributed) console.log(`\n${unattributed} ruling(s) came from a step this
|
|
|
310
310
|
}
|
|
311
311
|
}
|
|
312
312
|
console.log(`\nthe log now holds ${all.length} ruling(s) — simframe supervisions --device=${dev.udid}`);
|
|
313
|
+
|
|
314
|
+
// Close the helper, or this script never exits.
|
|
315
|
+
//
|
|
316
|
+
// The supervisor keeps a warm child process and deliberately does not unref its
|
|
317
|
+
// stdout — unreferencing it once unreferenced the pipe every request waits on,
|
|
318
|
+
// and the process then exited silently mid-await. The consequence nobody had
|
|
319
|
+
// noticed is at *this* end: the script finishes its work, prints this line, and
|
|
320
|
+
// then sits at 0% CPU forever with the helper idle at `readLine`, because the
|
|
321
|
+
// live child is holding the event loop open.
|
|
322
|
+
//
|
|
323
|
+
// That is the whole story of three "orphaned" collect-rulings processes killed
|
|
324
|
+
// by PID the night before, which were attributed to a task runner not killing
|
|
325
|
+
// its children. They had each finished their work and could not leave.
|
|
326
|
+
supervisor.close();
|
|
@@ -129,7 +129,24 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
129
129
|
if (screen.steps?.length) {
|
|
130
130
|
// Verification is off: this measures fingerprints, and a wrong-turn
|
|
131
131
|
// verdict computed from the very tokens under test would be circular.
|
|
132
|
-
|
|
132
|
+
//
|
|
133
|
+
// The tour asserts arrival rather than sleeping through it, so a step
|
|
134
|
+
// here can now fail — `openUrl` returns NSPOSIXErrorDomain 60 on a loaded
|
|
135
|
+
// hosted runner, and a page that never renders no longer passes silently
|
|
136
|
+
// as a six-token reading. That is the trade this harness wants: a loud
|
|
137
|
+
// failure naming the screen, over a quiet one that shows up thirty lines
|
|
138
|
+
// later as a threshold with no clearance. Everything read so far is
|
|
139
|
+
// written out first, because a failing run is the one whose evidence
|
|
140
|
+
// matters.
|
|
141
|
+
try {
|
|
142
|
+
await actions.runScript(device, { steps: screen.steps, verify: false });
|
|
143
|
+
} catch (err) {
|
|
144
|
+
save({ abandonedAt: { screen: screen.name, round, error: err.message } });
|
|
145
|
+
console.error(`\nFAIL round ${round}, "${screen.name}" never arrived: ${err.message}`);
|
|
146
|
+
console.error('The tour waits for something each screen actually shows. This is that wait');
|
|
147
|
+
console.error('giving up — not a fingerprint result. Check the app, the network, or the runner.');
|
|
148
|
+
process.exit(1);
|
|
149
|
+
}
|
|
133
150
|
}
|
|
134
151
|
const id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
135
152
|
// Did we actually arrive? Two differently-named screens reading the same
|
|
@@ -339,7 +356,16 @@ console.log(`${thresholdInGap ? 'ok ' : 'WARN'} the threshold ${thresholdInGap
|
|
|
339
356
|
const worstSame = [...same].sort((a, b) => a.similarity - b.similarity)[0];
|
|
340
357
|
const worstDifferent = [...different].sort((a, b) => b.similarity - a.similarity)[0];
|
|
341
358
|
if (worstSame) {
|
|
342
|
-
|
|
359
|
+
// Say whether either side was read off a screen that was still moving.
|
|
360
|
+
//
|
|
361
|
+
// The counts have always been printed and the settle flag never was, so a red
|
|
362
|
+
// run showing `11 vs 6 tokens` left it open whether the fingerprint had
|
|
363
|
+
// drifted or one reading had simply been taken too early. It is printed per
|
|
364
|
+
// reading above, thirty lines away and on a different row; here it is beside
|
|
365
|
+
// the number it explains.
|
|
366
|
+
const moving = [worstSame.a, worstSame.b].filter((r) => r.settled === false).length;
|
|
367
|
+
const movingNote = moving ? `, ${moving === 2 ? 'both readings' : 'one reading'} taken on a screen that never settled` : '';
|
|
368
|
+
console.log(`\nweakest same-screen pair: ${worstSame.a.name} r${worstSame.a.round} vs r${worstSame.b.round} = ${f(worstSame.similarity)} (${worstSame.a.count} vs ${worstSame.b.count} tokens${movingNote})`);
|
|
343
369
|
}
|
|
344
370
|
if (worstDifferent) {
|
|
345
371
|
console.log(`closest different-screen pair: ${worstDifferent.a.name} vs ${worstDifferent.b.name} = ${f(worstDifferent.similarity)}`);
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Ask several supervisors the same questions.
|
|
3
|
+
//
|
|
4
|
+
// node scripts/replay-rulings.mjs --device=<udid> --arms=apple,ollama:qwen3:8b,ollama:qwen3:14b
|
|
5
|
+
//
|
|
6
|
+
// The owner's call, 2026-09-11: settle the capacity question with numbers
|
|
7
|
+
// rather than speculation. This is how, and the shape matters more than the
|
|
8
|
+
// result.
|
|
9
|
+
//
|
|
10
|
+
// **Why replay rather than re-drive.** Collecting a population per arm means
|
|
11
|
+
// driving the simulator once per arm — about half an hour each — and, worse, it
|
|
12
|
+
// puts the *device's* variance inside a comparison that is supposed to be about
|
|
13
|
+
// the judges. A list that happened to arrive faster on one pass than another is
|
|
14
|
+
// not a fact about a model. So one device pass records the situations (see
|
|
15
|
+
// `situation` in metrics.recordSupervision) and every arm answers the identical
|
|
16
|
+
// set.
|
|
17
|
+
//
|
|
18
|
+
// **What replay cannot measure**, stated here rather than discovered later: the
|
|
19
|
+
// `outcome` column. Whether acting on a ruling actually recovered the flow is a
|
|
20
|
+
// fact about the device at that moment, and it belongs to the arm that was
|
|
21
|
+
// live. Replay scores *decisions*. The distinction is load-bearing — the live
|
|
22
|
+
// population scored 91% by decision and 69% by outcome, and almost all of that
|
|
23
|
+
// gap was one fixture that took the right word eight times and recovered once.
|
|
24
|
+
import * as metrics from '../src/metrics.js';
|
|
25
|
+
import * as ollama from '../src/ollama.js';
|
|
26
|
+
import * as supervisor from '../src/supervisor.js';
|
|
27
|
+
import { resolveDevice } from '../src/platform/index.js';
|
|
28
|
+
|
|
29
|
+
const arg = (n, d) => {
|
|
30
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
31
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
/** The fixtures and the word each situation actually calls for. Same table as the scorer. */
|
|
35
|
+
const CORRECT = {
|
|
36
|
+
'the list is still loading; its rows arrive shortly after launch': 'wait',
|
|
37
|
+
'the detail screen is still fetching; its text arrives shortly': 'wait',
|
|
38
|
+
'the list arrives in waves and this row is in the last one': 'wait',
|
|
39
|
+
'Review is blocked until Species is filled in, and it is empty': 'stop',
|
|
40
|
+
'the first submit always fails and the second works, so waiting cannot help': 'stop',
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
/** `retry` and `wait` differ only in how long they settle, so both satisfy a wait. */
|
|
44
|
+
const satisfies = (d, want) => (want === 'wait' ? d === 'wait' || d === 'retry' : d === want);
|
|
45
|
+
|
|
46
|
+
const dev = await resolveDevice(arg('device'));
|
|
47
|
+
const arms = String(arg('arms', 'apple')).split(',').map((a) => a.trim()).filter(Boolean);
|
|
48
|
+
const lastN = Number(arg('last', 0));
|
|
49
|
+
|
|
50
|
+
const all = metrics.readSupervisions(dev.udid)
|
|
51
|
+
.filter((r) => CORRECT[r.expect] && r.situation?.step && r.situation?.failure);
|
|
52
|
+
const rows = lastN > 0 ? all.slice(-lastN) : all;
|
|
53
|
+
|
|
54
|
+
if (rows.length < 8) {
|
|
55
|
+
console.error(`only ${rows.length} replayable ruling(s) — need situations recorded, which`);
|
|
56
|
+
console.error('means a population collected after the situation field was added.');
|
|
57
|
+
console.error('Run: SIMFRAME_SUPERVISOR=apple node scripts/collect-rulings.mjs --device=<udid>');
|
|
58
|
+
process.exit(2);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const want = (r) => CORRECT[r.expect];
|
|
62
|
+
const counts = rows.reduce((a, r) => ({ ...a, [want(r)]: (a[want(r)] ?? 0) + 1 }), {});
|
|
63
|
+
const commonest = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
|
|
64
|
+
const baseline = commonest[1] / rows.length;
|
|
65
|
+
const pct = (x) => `${(100 * x).toFixed(0)}%`;
|
|
66
|
+
|
|
67
|
+
console.log(`${rows.length} situations from ${dev.name}`);
|
|
68
|
+
console.log(`balance: ${JSON.stringify(counts)}`);
|
|
69
|
+
console.log(`majority-class baseline: always "${commonest[0]}" scores ${pct(baseline)}`);
|
|
70
|
+
if (baseline > 0.65) {
|
|
71
|
+
console.log('\nSKEWED — any accuracy below is mostly a fact about the fixture set.');
|
|
72
|
+
}
|
|
73
|
+
// The free comparison, computed here so it is in the same table as the models
|
|
74
|
+
// rather than in a different report. It is not an arm; it is the thing every
|
|
75
|
+
// arm has to beat to be worth its latency.
|
|
76
|
+
const STILL_MS_THRESHOLD = 3000;
|
|
77
|
+
console.log(`\nthe brief every model arm gets is ${ollama.readBrief().length} characters, read from native/supervise.swift`);
|
|
78
|
+
|
|
79
|
+
const results = [];
|
|
80
|
+
for (const arm of arms) {
|
|
81
|
+
const judged = [];
|
|
82
|
+
let unanswered = 0;
|
|
83
|
+
// Weights first, so the first situation is not timing a disk read.
|
|
84
|
+
if (arm.startsWith('ollama')) await ollama.preload(ollama.parseTarget(arm));
|
|
85
|
+
process.stdout.write(`\nasking ${arm} …`);
|
|
86
|
+
for (const r of rows) {
|
|
87
|
+
const detail = {};
|
|
88
|
+
const ruling = await supervisor.judge({
|
|
89
|
+
...r.situation,
|
|
90
|
+
options: { supervisor: arm },
|
|
91
|
+
// Wide on purpose. The shipped budget is 2.5s and a 14B will exceed it;
|
|
92
|
+
// capping here would score the larger model on *latency* while calling it
|
|
93
|
+
// accuracy, and latency is reported separately below where it can be read
|
|
94
|
+
// for what it is.
|
|
95
|
+
timeoutMs: 30_000,
|
|
96
|
+
detail,
|
|
97
|
+
});
|
|
98
|
+
if (!ruling) unanswered += 1;
|
|
99
|
+
judged.push({ r, ruling, detail });
|
|
100
|
+
process.stdout.write('.');
|
|
101
|
+
}
|
|
102
|
+
const answered = judged.filter((j) => j.ruling);
|
|
103
|
+
const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
|
|
104
|
+
const lat = answered.map((j) => j.ruling.ms).filter(Number.isFinite).sort((a, b) => a - b);
|
|
105
|
+
results.push({
|
|
106
|
+
arm,
|
|
107
|
+
n: rows.length,
|
|
108
|
+
unanswered,
|
|
109
|
+
// Scored over every situation, not only the answered ones. A judge that
|
|
110
|
+
// declines half the questions and is right about the rest is not an 100%
|
|
111
|
+
// judge, and scoring only its answers would say it was.
|
|
112
|
+
accuracy: right / rows.length,
|
|
113
|
+
medianMs: lat.length ? lat[lat.length >> 1] : null,
|
|
114
|
+
errors: answered
|
|
115
|
+
.filter((j) => !satisfies(j.ruling.decision, want(j.r)))
|
|
116
|
+
.map((j) => `said "${j.ruling.decision}" where "${want(j.r)}" was right — ${j.r.expect.slice(0, 48)}`),
|
|
117
|
+
});
|
|
118
|
+
process.stdout.write(' done\n');
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const stillRule = rows.filter((r) => {
|
|
122
|
+
const d = (r.situation.stillMs ?? 0) > STILL_MS_THRESHOLD ? 'stop' : 'wait';
|
|
123
|
+
return satisfies(d, want(r));
|
|
124
|
+
}).length / rows.length;
|
|
125
|
+
|
|
126
|
+
console.log(`\n${'arm'.padEnd(26)} ${'accuracy'.padStart(9)} ${'median'.padStart(9)} ${'no answer'.padStart(10)}`);
|
|
127
|
+
console.log('-'.repeat(58));
|
|
128
|
+
console.log(`${`always "${commonest[0]}"`.padEnd(26)} ${pct(baseline).padStart(9)} ${'—'.padStart(9)} ${'—'.padStart(10)}`);
|
|
129
|
+
console.log(`${`stillMs > ${STILL_MS_THRESHOLD}ms`.padEnd(26)} ${pct(stillRule).padStart(9)} ${'0ms'.padStart(9)} ${'—'.padStart(10)}`);
|
|
130
|
+
for (const r of results) {
|
|
131
|
+
console.log(`${r.arm.padEnd(26)} ${pct(r.accuracy).padStart(9)} ${`${r.medianMs ?? '—'}ms`.padStart(9)} ${`${r.unanswered}`.padStart(10)}`);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
console.log('\nerrors, by arm:');
|
|
135
|
+
for (const r of results) {
|
|
136
|
+
console.log(` ${r.arm}`);
|
|
137
|
+
if (!r.errors.length) console.log(' none');
|
|
138
|
+
const seen = new Map();
|
|
139
|
+
for (const e of r.errors) seen.set(e, (seen.get(e) ?? 0) + 1);
|
|
140
|
+
for (const [e, n] of [...seen].sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${e}`);
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
console.log('\nThis scores DECISIONS on identical inputs. It cannot score outcomes —');
|
|
144
|
+
console.log('whether acting on a ruling recovered the flow is a fact about the device at');
|
|
145
|
+
console.log('that moment, and belongs to whichever arm was live. See docs/EXPERIMENTS.md.');
|
|
146
|
+
|
|
147
|
+
supervisor.close();
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Score a ruling population against the baselines that could embarrass it.
|
|
3
|
+
//
|
|
4
|
+
// Written **before** the population it first scored was finished, deliberately.
|
|
5
|
+
// The previous attempt computed a `stillMs` threshold after seeing the answers,
|
|
6
|
+
// on 14 samples of which 12 shared one label, and produced "100%" — a number
|
|
7
|
+
// that was fitted, not measured. Fixing that afterwards is not possible: you
|
|
8
|
+
// cannot un-see the data. So the split rule, the candidate thresholds and the
|
|
9
|
+
// baselines are all fixed here in advance.
|
|
10
|
+
//
|
|
11
|
+
// node scripts/score-rulings.mjs --device=<udid>
|
|
12
|
+
//
|
|
13
|
+
// What it will not do: pick the best threshold and report its score. It fits on
|
|
14
|
+
// the first half and scores on the second, and prints both numbers so a gap
|
|
15
|
+
// between them is visible as overfitting rather than hidden as success.
|
|
16
|
+
import * as metrics from '../src/metrics.js';
|
|
17
|
+
import { resolveDevice } from '../src/platform/index.js';
|
|
18
|
+
|
|
19
|
+
const arg = (n, d) => {
|
|
20
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
21
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
22
|
+
};
|
|
23
|
+
|
|
24
|
+
/** The fixtures, and the word each one's situation actually calls for. */
|
|
25
|
+
const CORRECT = {
|
|
26
|
+
'the list is still loading; its rows arrive shortly after launch': 'wait',
|
|
27
|
+
'the detail screen is still fetching; its text arrives shortly': 'wait',
|
|
28
|
+
'the list arrives in waves and this row is in the last one': 'wait',
|
|
29
|
+
'Review is blocked until Species is filled in, and it is empty': 'stop',
|
|
30
|
+
'the first submit always fails and the second works, so waiting cannot help': 'stop',
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
/** `retry` and `wait` differ only in how long they settle, so both satisfy a wait. */
|
|
34
|
+
const satisfies = (decision, want) =>
|
|
35
|
+
(want === 'wait' ? decision === 'wait' || decision === 'retry' : decision === want);
|
|
36
|
+
|
|
37
|
+
const dev = await resolveDevice(arg('device'));
|
|
38
|
+
const all = metrics.readSupervisions(dev.udid).filter((r) => CORRECT[r.expect]);
|
|
39
|
+
// `--last=N` scores one batch rather than the whole log. Needed the first time
|
|
40
|
+
// this ran: the log still held rulings from a deliberately skewed population,
|
|
41
|
+
// so the balance read 19/35 and the baseline 65% when the batch actually under
|
|
42
|
+
// test was 16/16. Mixing a known-bad population into the denominator is the
|
|
43
|
+
// same error as before wearing a different hat.
|
|
44
|
+
const lastN = Number(arg('last', 0));
|
|
45
|
+
// Which judge. Every arm of the capacity comparison writes to one log, so
|
|
46
|
+
// scoring without this would average a 3B, an 8B and a 14B into a single
|
|
47
|
+
// meaningless number — and it would look like a result.
|
|
48
|
+
const arm = arg('arm', null);
|
|
49
|
+
const armOf = (r) => r.supervisor ?? 'unrecorded';
|
|
50
|
+
const scoped = arm ? all.filter((r) => armOf(r) === arm) : all;
|
|
51
|
+
const rows = lastN > 0 ? scoped.slice(-lastN) : scoped;
|
|
52
|
+
|
|
53
|
+
// A population spanning several arms is not one population. Say so, and say
|
|
54
|
+
// what to pass, rather than printing an average of things that were never
|
|
55
|
+
// comparable.
|
|
56
|
+
const arms = [...new Set(rows.map(armOf))];
|
|
57
|
+
if (arms.length > 1) {
|
|
58
|
+
console.log(`this log holds rulings from ${arms.length} supervisor arms:`);
|
|
59
|
+
for (const a of arms) console.log(` ${String(rows.filter((r) => armOf(r) === a).length).padStart(4)} ${a}`);
|
|
60
|
+
console.log('\nScore one at a time — `--arm=<name>` — because averaging them is not a result.');
|
|
61
|
+
process.exit(2);
|
|
62
|
+
}
|
|
63
|
+
if (rows.length) console.log(`arm: ${arms[0]}\n`);
|
|
64
|
+
if (rows.length < 8) {
|
|
65
|
+
console.error(`only ${rows.length} labelled ruling(s) — run scripts/collect-rulings.mjs first`);
|
|
66
|
+
process.exit(2);
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
const want = (r) => CORRECT[r.expect];
|
|
70
|
+
const counts = rows.reduce((a, r) => ({ ...a, [want(r)]: (a[want(r)] ?? 0) + 1 }), {});
|
|
71
|
+
const commonest = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
|
|
72
|
+
const baseline = commonest[1] / rows.length;
|
|
73
|
+
|
|
74
|
+
console.log(`${rows.length} labelled rulings on ${dev.name}`);
|
|
75
|
+
console.log(`balance: ${JSON.stringify(counts)}`);
|
|
76
|
+
console.log(`majority-class baseline: always "${commonest[0]}" scores ${(100 * baseline).toFixed(0)}%`);
|
|
77
|
+
if (baseline > 0.65) {
|
|
78
|
+
console.log('\nSKEWED — any accuracy below is mostly a fact about the fixture set.');
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
const acc = (rs, predict) => rs.filter((r) => satisfies(predict(r), want(r))).length / (rs.length || 1);
|
|
82
|
+
const pct = (x) => `${(100 * x).toFixed(0)}%`;
|
|
83
|
+
|
|
84
|
+
console.log(`\n${'the model'.padEnd(30)} ${pct(acc(rows, (r) => r.decision))}`);
|
|
85
|
+
console.log(`${`always "${commonest[0]}"`.padEnd(30)} ${pct(baseline)}`);
|
|
86
|
+
|
|
87
|
+
// --- the stillMs rule, fitted on one half and scored on the other.
|
|
88
|
+
//
|
|
89
|
+
// Split by *recording order* rather than at random or by fixture, because the
|
|
90
|
+
// alternative is choosing a split, and choosing is the thing that went wrong.
|
|
91
|
+
const half = Math.floor(rows.length / 2);
|
|
92
|
+
const fit = rows.slice(0, half);
|
|
93
|
+
const held = rows.slice(half);
|
|
94
|
+
const CANDIDATES = [1000, 1500, 2000, 2500, 3000, 4000, 5000, 6000, 7000];
|
|
95
|
+
const rule = (t) => (r) => ((r.still_ms ?? 0) > t ? 'stop' : 'wait');
|
|
96
|
+
|
|
97
|
+
let best = null;
|
|
98
|
+
for (const t of CANDIDATES) {
|
|
99
|
+
const a = acc(fit, rule(t));
|
|
100
|
+
if (!best || a > best.a) best = { t, a };
|
|
101
|
+
}
|
|
102
|
+
console.log(`\nstillMs threshold, fitted on the first ${fit.length} and scored on the last ${held.length}:`);
|
|
103
|
+
console.log(` chosen on the fit half: > ${best.t}ms (${pct(best.a)} there)`);
|
|
104
|
+
console.log(` ${'on the held-out half:'.padEnd(24)} ${pct(acc(held, rule(best.t)))}`);
|
|
105
|
+
console.log(` ${'the model, same half:'.padEnd(24)} ${pct(acc(held, (r) => r.decision))}`);
|
|
106
|
+
console.log('\nA rule that scores far better on the fit half than the held-out half was');
|
|
107
|
+
console.log('fitted to noise. That gap is the number this script exists to print.');
|
|
108
|
+
|
|
109
|
+
// --- the direction of the errors, which does not depend on the balance at all.
|
|
110
|
+
const wrong = rows.filter((r) => !satisfies(r.decision, want(r)));
|
|
111
|
+
const dirs = wrong.reduce((a, r) => {
|
|
112
|
+
const k = `said "${r.decision}" where "${want(r)}" was right`;
|
|
113
|
+
return { ...a, [k]: (a[k] ?? 0) + 1 };
|
|
114
|
+
}, {});
|
|
115
|
+
console.log(`\n${wrong.length} error(s) of ${rows.length}:`);
|
|
116
|
+
for (const [k, n] of Object.entries(dirs).sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${k}`);
|
|
117
|
+
if (Object.keys(dirs).length === 1 && wrong.length > 2) {
|
|
118
|
+
console.log('\nAll errors in one direction. A one-directional bias is what an abstain');
|
|
119
|
+
console.log('token addresses (DEFERRED 100), whatever the headline accuracy says.');
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
const lat = rows.map((r) => r.latency_ms).filter(Number.isFinite).sort((a, b) => a - b);
|
|
123
|
+
if (lat.length) console.log(`\nmedian judgement latency: ${lat[lat.length >> 1]}ms`);
|
|
124
|
+
const timed = rows.filter((r) => Number.isFinite(r.edge_p95_ms)).length;
|
|
125
|
+
console.log(`edges the graph had timed: ${timed}/${rows.length}`);
|
package/src/actions.js
CHANGED
|
@@ -260,6 +260,8 @@ export async function runScript(
|
|
|
260
260
|
expect,
|
|
261
261
|
failure,
|
|
262
262
|
outcome,
|
|
263
|
+
supervisor: supervisor.requested(options),
|
|
264
|
+
situation: ruling?.context?.situation ?? null,
|
|
263
265
|
});
|
|
264
266
|
} catch {
|
|
265
267
|
/* instrumentation must not be able to fail a flow it is only watching */
|
|
@@ -354,12 +356,19 @@ export async function runScript(
|
|
|
354
356
|
}),
|
|
355
357
|
observedMs: null,
|
|
356
358
|
};
|
|
359
|
+
// Where the hands actually went, filled in by the step that moved them.
|
|
360
|
+
//
|
|
361
|
+
// Item 120 needs the *resolved* point, not the one in the script: an
|
|
362
|
+
// image-space `tapAt` is converted inside the step, and `scroll` invents
|
|
363
|
+
// its coordinates from the screen size. Reading them back off the step
|
|
364
|
+
// would diagnose a point nothing was ever aimed at.
|
|
365
|
+
const aim = { at: null };
|
|
357
366
|
let detail;
|
|
358
367
|
// A selector that did not resolve gets the step's own alternatives before
|
|
359
368
|
// the batch is abandoned. Anything else propagates: retrying from a screen
|
|
360
369
|
// we did not expect to be on is not a retry, it is a second guess.
|
|
361
370
|
try {
|
|
362
|
-
detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
|
|
371
|
+
detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus, aim });
|
|
363
372
|
} catch (thrown) {
|
|
364
373
|
let err = thrown;
|
|
365
374
|
// Ask the supervisor before anything is abandoned. It sits behind the
|
|
@@ -393,7 +402,7 @@ export async function runScript(
|
|
|
393
402
|
options,
|
|
394
403
|
}).catch(() => null);
|
|
395
404
|
try {
|
|
396
|
-
detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus });
|
|
405
|
+
detail = await runStep(deviceQuery, udid, step, { screen, options, frames, focus, aim });
|
|
397
406
|
detail += ` [the local supervisor said ${ruling.decision}; it worked on the second attempt]`;
|
|
398
407
|
ruled('recovered');
|
|
399
408
|
continue;
|
|
@@ -433,7 +442,7 @@ export async function runScript(
|
|
|
433
442
|
let last = err;
|
|
434
443
|
for (const label of allowed) {
|
|
435
444
|
try {
|
|
436
|
-
detail = await runStep(deviceQuery, udid, stepWithTarget(step, label), { screen, options, frames, focus });
|
|
445
|
+
detail = await runStep(deviceQuery, udid, stepWithTarget(step, label), { screen, options, frames, focus, aim });
|
|
437
446
|
detail += ` [after ${tried.map((t) => JSON.stringify(String(t))).join(', ')} did not resolve]`;
|
|
438
447
|
last = null;
|
|
439
448
|
break;
|
|
@@ -729,10 +738,40 @@ export async function runScript(
|
|
|
729
738
|
? ' [the screen did not change, so this app was already in front — or it did not come forward]'
|
|
730
739
|
: '';
|
|
731
740
|
const filling = stillFillingIn(afterReading?.entry);
|
|
741
|
+
// Item 120: when a gesture aimed at a coordinate does nothing, say what
|
|
742
|
+
// that coordinate resolved to. The element that swallowed it is normally
|
|
743
|
+
// already in the list printed under this very verdict — the gap was
|
|
744
|
+
// never the data, it was that nobody connected "started at y=750" to
|
|
745
|
+
// "there is a banner at y=753".
|
|
746
|
+
//
|
|
747
|
+
// Diagnosed against the screen as it was *before* the action, because
|
|
748
|
+
// that is the screen the finger landed on. Using the after-reading would
|
|
749
|
+
// describe the world the gesture failed to change.
|
|
750
|
+
// Both signals, because they are not the same one and only one of them
|
|
751
|
+
// fires in the reported case. `settled.noVisibleChange` is the pixel
|
|
752
|
+
// detector saying the frame never moved; the `no-visible-change` verdict
|
|
753
|
+
// is the fingerprint saying we are on the screen we started on. A swipe
|
|
754
|
+
// into a search bar settled in 62ms and produced the second without the
|
|
755
|
+
// first — and six swipes reporting "no visible change" is what item 120
|
|
756
|
+
// was reported against. Keying on one of them would have shipped a fix
|
|
757
|
+
// that did not fire on its own bug report.
|
|
758
|
+
const wentNowhere = Boolean(settled?.noVisibleChange) || verification?.verdict === 'no-visible-change';
|
|
759
|
+
const aimNote = wentNowhere && aim.at
|
|
760
|
+
? screenmap.describePoint(
|
|
761
|
+
// `beforeScreen` is null with verification off, and that is exactly
|
|
762
|
+
// the mode someone falls back to when coordinates are misbehaving.
|
|
763
|
+
// A stored map keyed on the pre-action frame costs one file read.
|
|
764
|
+
beforeScreen?.entry ?? screenmap.recall(udid, before),
|
|
765
|
+
aim.at.point,
|
|
766
|
+
{ what: aim.at.what },
|
|
767
|
+
)
|
|
768
|
+
: null;
|
|
732
769
|
const note = launchNote
|
|
733
770
|
+ (filling ? ` [settled, but ${filling} — waitFor content, do not act on this yet]` : '')
|
|
734
771
|
+ (settled?.smallChange ? ' [a small change, in one region only]' : '')
|
|
735
772
|
+ (settled?.noVisibleChange ? ' [no visible change]' : '')
|
|
773
|
+
// After the symptom, because it is the explanation of it.
|
|
774
|
+
+ (aimNote ? ` [${aimNote}]` : '')
|
|
736
775
|
+ (settled?.staleBaseline ? ' [baseline had already settled; re-taken from the live screen]' : '')
|
|
737
776
|
+ (settled?.blackFrames
|
|
738
777
|
? ` [${settled.blackFrames} black frame(s) waited through${settled.blackMs ? `, still black after ${settled.blackMs}ms` : ''}]`
|
|
@@ -1215,17 +1254,25 @@ async function superviseFailure(deviceQuery, { goal, step, expected, err, option
|
|
|
1215
1254
|
try {
|
|
1216
1255
|
timing = udid && map.identity ? graph.timingFor(udid, map.identity, step) : null;
|
|
1217
1256
|
} catch { /* an unknown edge is a fact about the graph, not a failure here */ }
|
|
1218
|
-
|
|
1219
|
-
|
|
1257
|
+
// The question, kept — not just the answer.
|
|
1258
|
+
//
|
|
1259
|
+
// Three items (101, 96, 106) want to ask "what would a different judge have
|
|
1260
|
+
// said about these same situations", and until now the log held the verdict
|
|
1261
|
+
// and threw away what was asked. So comparing two supervisors meant driving
|
|
1262
|
+
// the device once per arm, which is thirty minutes an arm and introduces
|
|
1263
|
+
// the device's own variance into a comparison that is supposed to be about
|
|
1264
|
+
// the judges. With this, one device pass produces a population every arm
|
|
1265
|
+
// can be asked, on identical inputs.
|
|
1266
|
+
const situation = {
|
|
1267
|
+
goal: goal ?? null,
|
|
1220
1268
|
step: `${step.action} ${JSON.stringify(String(step.value ?? step.target ?? step.into ?? step.seek ?? '').slice(0, 60))}`,
|
|
1221
|
-
expected,
|
|
1269
|
+
expected: expected ?? null,
|
|
1222
1270
|
failure: err.message,
|
|
1223
1271
|
screen: (map.rows ?? []).filter((r) => r.label).map((r) => r.label),
|
|
1224
1272
|
stillMs,
|
|
1225
1273
|
note: stillFillingIn(map.identity?.entry),
|
|
1226
|
-
|
|
1227
|
-
|
|
1228
|
-
});
|
|
1274
|
+
};
|
|
1275
|
+
const ruling = await supervisor.judge({ ...situation, options, detail });
|
|
1229
1276
|
if (!ruling) return null;
|
|
1230
1277
|
return {
|
|
1231
1278
|
...ruling,
|
|
@@ -1237,6 +1284,7 @@ async function superviseFailure(deviceQuery, { goal, step, expected, err, option
|
|
|
1237
1284
|
stillMs,
|
|
1238
1285
|
p95: timing?.p95 ?? null,
|
|
1239
1286
|
samples: timing?.samples ?? null,
|
|
1287
|
+
situation,
|
|
1240
1288
|
},
|
|
1241
1289
|
};
|
|
1242
1290
|
} catch {
|
|
@@ -2166,6 +2214,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2166
2214
|
pointHeight: geo.pointHeight,
|
|
2167
2215
|
}));
|
|
2168
2216
|
}
|
|
2217
|
+
if (ctx.aim) ctx.aim.at = { point: { x, y }, what: 'the tap point' };
|
|
2169
2218
|
await input.tapPoint(udid, x, y, { durationMs: step.durationMs });
|
|
2170
2219
|
return `tapped ${x},${y}`;
|
|
2171
2220
|
}
|
|
@@ -2253,6 +2302,9 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2253
2302
|
case 'swipe': {
|
|
2254
2303
|
const from = { x: step.from?.[0] ?? step.from?.x, y: step.from?.[1] ?? step.from?.y };
|
|
2255
2304
|
const to = { x: step.to?.[0] ?? step.to?.x, y: step.to?.[1] ?? step.to?.y };
|
|
2305
|
+
// The start point only. A swipe is captured by whatever the finger goes
|
|
2306
|
+
// down on; where it lifts never decides who received it.
|
|
2307
|
+
if (ctx.aim) ctx.aim.at = { point: from, what: 'the swipe start point' };
|
|
2256
2308
|
await input.swipe(udid, from, to, { durationMs: step.durationMs });
|
|
2257
2309
|
return `swiped ${from.x},${from.y} -> ${to.x},${to.y}`;
|
|
2258
2310
|
}
|
|
@@ -2269,6 +2321,7 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2269
2321
|
right: [{ x: midX - span, y: midY }, { x: midX + span, y: midY }],
|
|
2270
2322
|
};
|
|
2271
2323
|
if (!moves[dir]) throw new Error(`unknown scroll direction "${dir}"`);
|
|
2324
|
+
if (ctx.aim) ctx.aim.at = { point: moves[dir][0], what: 'the scroll start point' };
|
|
2272
2325
|
await input.swipe(udid, moves[dir][0], moves[dir][1], { durationMs: step.durationMs ?? 250 });
|
|
2273
2326
|
return `scrolled ${dir}`;
|
|
2274
2327
|
}
|
package/src/cli.js
CHANGED
|
@@ -278,6 +278,10 @@ async function main() {
|
|
|
278
278
|
if (flags.sensor) options.sensor = String(flags.sensor);
|
|
279
279
|
if (flags.planner) options.planner = String(flags.planner);
|
|
280
280
|
if (flags.supervisor) options.supervisor = String(flags.supervisor);
|
|
281
|
+
// Stated rather than defaulted, which is the same rule the settle budgets
|
|
282
|
+
// follow. The default is sized on a developer's machine; a loaded build farm
|
|
283
|
+
// is a different machine and should say so rather than be guessed at.
|
|
284
|
+
if (flags.readyTimeoutMs) options.readyTimeoutMs = num(flags.readyTimeoutMs);
|
|
281
285
|
|
|
282
286
|
switch (command) {
|
|
283
287
|
case undefined:
|
|
@@ -734,12 +738,25 @@ async function main() {
|
|
|
734
738
|
`"${target}" matches more than one screen:`,
|
|
735
739
|
...(res.candidates ?? []).map((c) => ` ${c.name} (${c.hash.slice(0, 8)})`),
|
|
736
740
|
],
|
|
741
|
+
// Both of these walked, so the steps they took are the useful part and
|
|
742
|
+
// are printed exactly as a flow prints them.
|
|
743
|
+
'route-halted': () => [
|
|
744
|
+
...(res.results ?? []).map(stepLine),
|
|
745
|
+
`stopped after ${res.ranSteps} of ${res.steps?.length} step(s) on the way to ${res.screen}`,
|
|
746
|
+
],
|
|
747
|
+
'arrived-elsewhere': () => [
|
|
748
|
+
...(res.results ?? []).map(stepLine),
|
|
749
|
+
`ended at ${res.arrived}, wanted ${res.screen} — every step ran, so an edge the graph`
|
|
750
|
+
+ ' remembers no longer leads where it says. Re-walk it and the graph will relearn.',
|
|
751
|
+
],
|
|
737
752
|
};
|
|
738
753
|
if (!res.ok && res.reason) {
|
|
739
754
|
emit(flags, res, refusal[res.reason] ?? `${res.reason}: cannot reach "${res.to ?? target}" from here`);
|
|
740
755
|
process.exitCode = 1;
|
|
741
756
|
return;
|
|
742
757
|
}
|
|
758
|
+
// Everything that is not ok now carries a reason and was handled above,
|
|
759
|
+
// so this is the arrival path only.
|
|
743
760
|
emit(
|
|
744
761
|
flags,
|
|
745
762
|
res,
|
|
@@ -747,9 +764,7 @@ async function main() {
|
|
|
747
764
|
? `already on ${res.screen}`
|
|
748
765
|
: [
|
|
749
766
|
...(res.results ?? []).map(stepLine),
|
|
750
|
-
res.
|
|
751
|
-
? `arrived at ${res.screen} in ${res.ranSteps} step(s)`
|
|
752
|
-
: `ended at ${res.arrived}, wanted ${res.screen}`,
|
|
767
|
+
`arrived at ${res.screen} in ${res.ranSteps} step(s)`,
|
|
753
768
|
],
|
|
754
769
|
);
|
|
755
770
|
process.exitCode = res.ok ? 0 : 1;
|
|
@@ -841,11 +856,15 @@ async function main() {
|
|
|
841
856
|
/* capture not running: fall through and report the send alone */
|
|
842
857
|
}
|
|
843
858
|
const t0 = Date.now();
|
|
859
|
+
// Item 120, the single-shot half: a coordinate that changed nothing owes
|
|
860
|
+
// an answer about what it landed on.
|
|
861
|
+
let aim = null;
|
|
844
862
|
switch (command) {
|
|
845
863
|
case 'tapAt': {
|
|
846
864
|
if (positional.length < 2 || nums.slice(0, 2).some(Number.isNaN)) {
|
|
847
865
|
throw new Error('usage: simframe tapAt <x> <y>');
|
|
848
866
|
}
|
|
867
|
+
aim = { point: { x: nums[0], y: nums[1] }, what: 'the tap point' };
|
|
849
868
|
await input.tapPoint(dev.udid, nums[0], nums[1], flags.durationMs ? { durationMs: num(flags.durationMs) } : {});
|
|
850
869
|
break;
|
|
851
870
|
}
|
|
@@ -853,6 +872,7 @@ async function main() {
|
|
|
853
872
|
if (positional.length < 4 || nums.slice(0, 4).some(Number.isNaN)) {
|
|
854
873
|
throw new Error('usage: simframe swipe <x1> <y1> <x2> <y2>');
|
|
855
874
|
}
|
|
875
|
+
aim = { point: { x: nums[0], y: nums[1] }, what: 'the swipe start point' };
|
|
856
876
|
await input.swipe(dev.udid, { x: nums[0], y: nums[1] }, { x: nums[2], y: nums[3] }, { durationMs: num(flags.durationMs, 300) });
|
|
857
877
|
break;
|
|
858
878
|
}
|
|
@@ -886,14 +906,22 @@ async function main() {
|
|
|
886
906
|
// Deliberately not an accusation. Pressing home while already on the
|
|
887
907
|
// springboard legitimately changes nothing, and a warning that cries wolf
|
|
888
908
|
// is how a real one gets ignored.
|
|
909
|
+
// The map for the screen as it was before the send. Free when the screen
|
|
910
|
+
// has been perceived once — and silent when it has not, because "we did
|
|
911
|
+
// not look" must not be printed as "there is nothing there".
|
|
912
|
+
const hit = changed === false && aim
|
|
913
|
+
? api.screenmap.describePoint(api.screenmap.recall(dev.udid, before), aim.point, { what: aim.what })
|
|
914
|
+
: null;
|
|
889
915
|
const note = changed === false
|
|
890
|
-
? ' — the screen did not change
|
|
916
|
+
? ' — the screen did not change'
|
|
917
|
+
+ (hit ? `, and ${hit}` : '')
|
|
918
|
+
+ '. That is expected if the press had nothing to do here;'
|
|
891
919
|
+ ' if you expected a change, input may not be reaching the device —'
|
|
892
920
|
+ ' `simframe stop --force && simframe start` rebuilds the session.'
|
|
893
921
|
: '';
|
|
894
922
|
emit(
|
|
895
923
|
flags,
|
|
896
|
-
{ ok: true, command, ms, driver: driver.name, screenChanged: changed },
|
|
924
|
+
{ ok: true, command, ms, driver: driver.name, screenChanged: changed, hit: hit ?? undefined },
|
|
897
925
|
`${command} in ${ms}ms via ${driver.name}${changed === true ? ' — screen changed' : ''}${note}`,
|
|
898
926
|
);
|
|
899
927
|
return;
|
package/src/index.js
CHANGED
|
@@ -110,6 +110,21 @@ async function deviceGeometry(udid, state) {
|
|
|
110
110
|
};
|
|
111
111
|
}
|
|
112
112
|
|
|
113
|
+
/**
|
|
114
|
+
* How long to wait for a daemon to exist at all. A first build is slower than a
|
|
115
|
+
* spawn, which is what this number is sized for.
|
|
116
|
+
*/
|
|
117
|
+
export const READY_TIMEOUT_MS = 20_000;
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* How long to keep waiting once the daemon is demonstrably alive.
|
|
121
|
+
*
|
|
122
|
+
* Sized from the measurement that caused it: a hosted runner's simulator took
|
|
123
|
+
* roughly 27 s to produce its first frame while the daemon reported a healthy
|
|
124
|
+
* 75 ms median. Three times the ordinary budget, and still bounded.
|
|
125
|
+
*/
|
|
126
|
+
export const LIVE_DAEMON_CAP_MS = 60_000;
|
|
127
|
+
|
|
113
128
|
export function daemonStatus(udid) {
|
|
114
129
|
const meta = store.readJson(store.paths(udid).meta);
|
|
115
130
|
const pid = meta?.pid ?? null;
|
|
@@ -146,16 +161,40 @@ export async function ensureDaemon(deviceQuery, options = {}) {
|
|
|
146
161
|
}
|
|
147
162
|
|
|
148
163
|
// The daemon may need a first build, which is slower than a spawn.
|
|
149
|
-
const deadline = Date.now() + (options.readyTimeoutMs ??
|
|
150
|
-
|
|
164
|
+
const deadline = Date.now() + (options.readyTimeoutMs ?? READY_TIMEOUT_MS);
|
|
165
|
+
// A live daemon that has not rendered yet is not a failed daemon.
|
|
166
|
+
//
|
|
167
|
+
// Measured on a hosted runner: `simframe start` gave up, and the daemon's own
|
|
168
|
+
// log — captured by the on-failure step eight seconds later — showed
|
|
169
|
+
// `frame=#1 age=1514ms, 1.0 fps, median 75.08ms`. It was working. The
|
|
170
|
+
// simulator's display had simply taken about 27 s to produce anything on a
|
|
171
|
+
// loaded build farm, against a 20 s budget measured on a developer's machine.
|
|
172
|
+
// Aborting there fails an entire run over a device that was about to work.
|
|
173
|
+
//
|
|
174
|
+
// So the budget applies to *getting a daemon*; once we have a live one, the
|
|
175
|
+
// wait extends to a hard cap. The cap still exists, because a daemon that is
|
|
176
|
+
// alive and never renders is a real failure and has to be reportable — but it
|
|
177
|
+
// is now a different sentence from a daemon that never started, and those two
|
|
178
|
+
// had read identically, which cost a log dive to tell apart.
|
|
179
|
+
const liveDeadline = Date.now() + LIVE_DAEMON_CAP_MS;
|
|
180
|
+
let sawLiveDaemon = false;
|
|
181
|
+
for (;;) {
|
|
151
182
|
const state = store.readJson(p.state);
|
|
152
183
|
if (state && state.capturedAt >= minCapturedAt && Date.now() - state.capturedAt < 30_000) {
|
|
153
184
|
return { device, state, started: !existing.alive };
|
|
154
185
|
}
|
|
186
|
+
const alive = daemonStatus(device.udid).alive;
|
|
187
|
+
sawLiveDaemon = sawLiveDaemon || alive;
|
|
188
|
+
if (Date.now() >= (alive ? liveDeadline : deadline)) break;
|
|
155
189
|
await sleep(80);
|
|
156
190
|
}
|
|
157
191
|
const tail = readLogTail(p.log);
|
|
158
|
-
|
|
192
|
+
const waited = sawLiveDaemon
|
|
193
|
+
? `the daemon is running and the display produced no frame in ${Math.round(LIVE_DAEMON_CAP_MS / 1000)}s`
|
|
194
|
+
: 'no daemon process came up';
|
|
195
|
+
throw new Error(
|
|
196
|
+
`simframe daemon did not produce a frame for ${device.name} — ${waited}${tail ? `\n${tail}` : ''}`,
|
|
197
|
+
);
|
|
159
198
|
}
|
|
160
199
|
|
|
161
200
|
/** Why the daemon was not used, when it was not. Surfaced by doctor. */
|
package/src/metrics.js
CHANGED
|
@@ -151,7 +151,7 @@ export const readSupervisions = (udid, opts) => readJsonl(metricPaths(udid).supe
|
|
|
151
151
|
*/
|
|
152
152
|
export function recordSupervision(udid, {
|
|
153
153
|
session, index, step, edge, screen, decision, from, reason, ms,
|
|
154
|
-
stillMs, p95, samples, expect, failure, outcome,
|
|
154
|
+
stillMs, p95, samples, expect, failure, outcome, supervisor, situation,
|
|
155
155
|
}) {
|
|
156
156
|
// Swallowed rather than thrown, unlike `recordEscalation`'s guard, and the
|
|
157
157
|
// difference is deliberate: this is called from inside a flow's failure
|
|
@@ -171,6 +171,19 @@ export function recordSupervision(udid, {
|
|
|
171
171
|
screen_fingerprint: screen ?? null,
|
|
172
172
|
decision,
|
|
173
173
|
from: from ?? 'model',
|
|
174
|
+
// *Which* judge, not just that there was one.
|
|
175
|
+
//
|
|
176
|
+
// Every arm of the capacity comparison writes to this one log, and without
|
|
177
|
+
// this field a population collected under Apple and one collected under a
|
|
178
|
+
// 14B are one undifferentiated file — the comparison the owner asked for
|
|
179
|
+
// would be unreadable from its own data. `from` says rule-or-model; this
|
|
180
|
+
// says which model.
|
|
181
|
+
supervisor: supervisor ?? null,
|
|
182
|
+
// The question, not only the answer. Without it, asking a second judge
|
|
183
|
+
// about the same situations means driving the device a second time — which
|
|
184
|
+
// puts the device's own variance inside a comparison that is about the
|
|
185
|
+
// judges. Null for a rule-sourced ruling, which never composed one.
|
|
186
|
+
situation: situation ?? null,
|
|
174
187
|
// Recorded, never presented as the ground for what happened: the supervisor
|
|
175
188
|
// has returned a correct decision with a reason citing a rule that did not
|
|
176
189
|
// apply. Keeping it is how that stays measurable instead of anecdotal.
|
|
@@ -312,6 +325,11 @@ export const PLAN_REASONS = {
|
|
|
312
325
|
'no-route': 'no_plan',
|
|
313
326
|
'unreplayable-edge': 'no_plan',
|
|
314
327
|
'unknown-flow': 'no_plan',
|
|
328
|
+
// A route that ran and did not land. Not `no_plan`: there *was* a plan and it
|
|
329
|
+
// was followed — what could not be confirmed is that it worked, which is what
|
|
330
|
+
// `verification_failed` means everywhere else in this file.
|
|
331
|
+
'route-halted': 'verification_failed',
|
|
332
|
+
'arrived-elsewhere': 'verification_failed',
|
|
315
333
|
};
|
|
316
334
|
|
|
317
335
|
/** A short, bounded description of a candidate element, for the log. */
|
package/src/navigate.js
CHANGED
|
@@ -96,14 +96,40 @@ export async function goto(deviceQuery, target, { options, ...runOptions } = {})
|
|
|
96
96
|
|
|
97
97
|
const result = await runScript(udid, { steps, stopOnUnexpected: true, ...runOptions });
|
|
98
98
|
const arrived = await api.screenIdentity(udid, {});
|
|
99
|
-
|
|
100
|
-
ok: arrived.hash === found.node.hash,
|
|
99
|
+
const walk = {
|
|
101
100
|
screen: found.name,
|
|
102
101
|
steps,
|
|
103
102
|
ranSteps: result.ranSteps,
|
|
104
103
|
results: result.results,
|
|
105
104
|
arrived: arrived.hash ? arrived.hash.slice(0, 8) : null,
|
|
106
105
|
};
|
|
106
|
+
if (arrived.hash === found.node.hash) return { ok: true, ...walk };
|
|
107
|
+
|
|
108
|
+
// A walk that ran and did not land had no name, and it was the only outcome
|
|
109
|
+
// here that did not.
|
|
110
|
+
//
|
|
111
|
+
// Every refusal above is named, logged and groupable; this one returned
|
|
112
|
+
// `{ok: false}` with no `reason` at all, so `simframe goto <known hash>` on
|
|
113
|
+
// CI failed the check that says *"it either walks there or names why it
|
|
114
|
+
// cannot"* with an empty detail — which is exactly what the check is for, and
|
|
115
|
+
// it had been sitting under an outcome nobody had named rather than under a
|
|
116
|
+
// crash.
|
|
117
|
+
//
|
|
118
|
+
// Two names, because they are two different faults and only one of them is
|
|
119
|
+
// about the graph. `route-halted` means a step on the route failed, which is
|
|
120
|
+
// an ordinary failure of the app or the moment. `arrived-elsewhere` means
|
|
121
|
+
// every step ran and we are *not where the graph promised* — an edge it
|
|
122
|
+
// remembers is wrong, and that is worth reading as a claim about memory
|
|
123
|
+
// rather than about this attempt.
|
|
124
|
+
const reason = !arrived.hash
|
|
125
|
+
? 'no-identity'
|
|
126
|
+
: (result.ranSteps < steps.length ? 'route-halted' : 'arrived-elsewhere');
|
|
127
|
+
return refuse(udid, { ok: false, reason, to: found.name, ...walk }, {
|
|
128
|
+
detail: reason === 'arrived-elsewhere'
|
|
129
|
+
? `route to "${found.name}" ran to the end and landed on ${walk.arrived}`
|
|
130
|
+
: `route to "${found.name}" stopped after ${result.ranSteps} of ${steps.length} step(s)`,
|
|
131
|
+
flowName: `goto:${target}`,
|
|
132
|
+
});
|
|
107
133
|
}
|
|
108
134
|
|
|
109
135
|
export function knownScreens(udid) {
|
package/src/ollama.js
ADDED
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A second supervisor arm, for measuring whether capacity matters.
|
|
3
|
+
*
|
|
4
|
+
* The owner's call, 2026-09-11: *"we can of course do our own test with a
|
|
5
|
+
* chosen model… and decide based on the numbers rather than speculations."*
|
|
6
|
+
* This exists to answer that and nothing else. It is **not** a recommendation,
|
|
7
|
+
* it is off unless asked for by name, and a model used to measure whether
|
|
8
|
+
* capacity matters is not a model we ship — conflating those is how a non-goal
|
|
9
|
+
* erodes.
|
|
10
|
+
*
|
|
11
|
+
* No dependency is added: Node has had a global `fetch` since 18, and Ollama
|
|
12
|
+
* is a process already on the machine, reached over the loopback interface.
|
|
13
|
+
*
|
|
14
|
+
* **The fairness condition, which is the whole reason this file is shaped the
|
|
15
|
+
* way it is.** Most small-model errors are invalid-output faults, and Apple's
|
|
16
|
+
* guided generation eliminates those at the sampling layer — constraints are
|
|
17
|
+
* enforced by logit masking, so a fourth word is unrepresentable rather than
|
|
18
|
+
* rejected afterwards (WWDC25 301; Tech Report §7). An unconstrained challenger
|
|
19
|
+
* would lose on formatting and we would read it as losing on judgement. So this
|
|
20
|
+
* arm passes a JSON schema whose `decision` is an enum of the same three words,
|
|
21
|
+
* and Ollama constrains sampling to it the same way.
|
|
22
|
+
*
|
|
23
|
+
* And the briefing is not a second copy. It is **read out of
|
|
24
|
+
* `native/supervise.swift`**, because two hand-maintained copies of a prompt is
|
|
25
|
+
* two arms answering different questions, and the difference would be invisible
|
|
26
|
+
* in the numbers. See `readBrief`.
|
|
27
|
+
*/
|
|
28
|
+
import fs from 'node:fs';
|
|
29
|
+
import path from 'node:path';
|
|
30
|
+
import { fileURLToPath } from 'node:url';
|
|
31
|
+
|
|
32
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
33
|
+
const SWIFT = path.join(HERE, '..', 'native', 'supervise.swift');
|
|
34
|
+
|
|
35
|
+
export const DEFAULT_MODEL = 'qwen3:8b';
|
|
36
|
+
export const DEFAULT_HOST = 'http://127.0.0.1:11434';
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* The Apple arm's own instructions, read from its source.
|
|
40
|
+
*
|
|
41
|
+
* Not duplicated. Two arms of a comparison that are briefed differently are
|
|
42
|
+
* measuring two different things, and nothing in the output would say so — the
|
|
43
|
+
* numbers would simply be wrong and look fine. The Swift file is the canonical
|
|
44
|
+
* copy because it is the shipped one; this parses the `let instructions = """
|
|
45
|
+
* … """` block out of it and un-escapes Swift's trailing-backslash line
|
|
46
|
+
* continuations, which is how that literal is wrapped.
|
|
47
|
+
*
|
|
48
|
+
* Throws rather than falling back to a built-in string. A silent fallback here
|
|
49
|
+
* would produce exactly the invisible unfairness this function exists to
|
|
50
|
+
* prevent.
|
|
51
|
+
*/
|
|
52
|
+
export function readBrief(file = SWIFT) {
|
|
53
|
+
const src = fs.readFileSync(file, 'utf8');
|
|
54
|
+
const open = src.indexOf('let instructions = """');
|
|
55
|
+
if (open === -1) throw new Error(`no instructions block in ${file}`);
|
|
56
|
+
const bodyStart = src.indexOf('\n', open) + 1;
|
|
57
|
+
const close = src.indexOf('"""', bodyStart);
|
|
58
|
+
if (close === -1) throw new Error(`unterminated instructions block in ${file}`);
|
|
59
|
+
return src
|
|
60
|
+
.slice(bodyStart, close)
|
|
61
|
+
.split('\n')
|
|
62
|
+
// A Swift multi-line literal strips the closing delimiter's indentation
|
|
63
|
+
// from every line; here that is four spaces.
|
|
64
|
+
.map((l) => l.replace(/^ {4}/, ''))
|
|
65
|
+
.join('\n')
|
|
66
|
+
// Swift's line continuation: a trailing backslash removes the newline and
|
|
67
|
+
// nothing else. Joining with a space instead would insert one after the
|
|
68
|
+
// space the line already ends with, and the two arms would be reading
|
|
69
|
+
// briefs that differ — invisibly, and in the one place this whole file
|
|
70
|
+
// exists to keep identical.
|
|
71
|
+
.replace(/\\\n/g, '')
|
|
72
|
+
.trim();
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
const clip = (t, n) => {
|
|
76
|
+
const s = String(t ?? '');
|
|
77
|
+
return s.length > n ? `${s.slice(0, n)}…` : s;
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* The situation, in the words the other arm gets.
|
|
82
|
+
*
|
|
83
|
+
* A line-for-line mirror of `main.swift`'s prompt assembly, including the clip
|
|
84
|
+
* lengths and the 14-label cap on the screen list. The field order matters as
|
|
85
|
+
* much as the content: the brief tells the model to weigh the plan's guidance
|
|
86
|
+
* first, and a prompt that presented it last would be testing a different
|
|
87
|
+
* instruction.
|
|
88
|
+
*/
|
|
89
|
+
export function promptFor(s = {}) {
|
|
90
|
+
let prompt = `Step: ${clip(s.step, 120)}\nIt failed with: ${clip(s.failure, 220)}`;
|
|
91
|
+
if (s.goal) prompt += `\nThe plan's guidance about this app: ${clip(s.goal, 300)}`;
|
|
92
|
+
if (s.expected) prompt += `\nExpected: ${clip(s.expected, 200)}`;
|
|
93
|
+
if (Number.isFinite(s.stillMs)) prompt += `\nThe screen has been still for ${s.stillMs}ms`;
|
|
94
|
+
if (s.note) prompt += `\nPerception note: ${clip(s.note, 160)}`;
|
|
95
|
+
if (s.screen?.length) {
|
|
96
|
+
prompt += `\nOn screen now: ${s.screen.slice(0, 14).map((l) => clip(l, 28)).join(', ')}`;
|
|
97
|
+
}
|
|
98
|
+
return prompt;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* The schema. Three words and nothing else, enforced at sampling.
|
|
103
|
+
*
|
|
104
|
+
* Deliberately decision-only, matching `Judgement` in the Swift file: that
|
|
105
|
+
* struct has one field. Asking this arm for a rationale it would then be scored
|
|
106
|
+
* against would be a second difference between the arms, and the rationale is
|
|
107
|
+
* the one thing a supervisor is explicitly not trusted for.
|
|
108
|
+
*/
|
|
109
|
+
export const SCHEMA = {
|
|
110
|
+
type: 'object',
|
|
111
|
+
properties: { decision: { type: 'string', enum: ['wait', 'retry', 'stop'] } },
|
|
112
|
+
required: ['decision'],
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
/** Parse `ollama`, `ollama:qwen3:14b`, `ollama:qwen3:8b@http://host:port`. */
|
|
116
|
+
export function parseTarget(raw) {
|
|
117
|
+
const rest = String(raw ?? '').replace(/^ollama:?/, '');
|
|
118
|
+
const [model, host] = rest.split('@');
|
|
119
|
+
return { model: model || DEFAULT_MODEL, host: host || DEFAULT_HOST };
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
async function post(host, route, body, timeoutMs) {
|
|
123
|
+
const control = new AbortController();
|
|
124
|
+
const timer = setTimeout(() => control.abort(), timeoutMs);
|
|
125
|
+
try {
|
|
126
|
+
const res = await fetch(`${host}${route}`, {
|
|
127
|
+
method: 'POST',
|
|
128
|
+
headers: { 'content-type': 'application/json' },
|
|
129
|
+
body: JSON.stringify(body),
|
|
130
|
+
signal: control.signal,
|
|
131
|
+
});
|
|
132
|
+
if (!res.ok) return { kind: 'http', error: `${res.status} ${await res.text().catch(() => '')}`.slice(0, 200) };
|
|
133
|
+
return { json: await res.json() };
|
|
134
|
+
} catch (err) {
|
|
135
|
+
return { kind: err.name === 'AbortError' ? 'timeout' : 'unreachable', error: String(err.message).slice(0, 200) };
|
|
136
|
+
} finally {
|
|
137
|
+
clearTimeout(timer);
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* One judgement. Same shape as the Apple helper's answer, so `judge` in
|
|
143
|
+
* `supervisor.js` cannot tell which arm it asked — including the failure
|
|
144
|
+
* shapes, because "the supervisor did not answer" covering a timeout, a
|
|
145
|
+
* refusal and a model that was never installed is a bug this project has
|
|
146
|
+
* already paid for once.
|
|
147
|
+
*/
|
|
148
|
+
export async function ask(target, situation, timeoutMs = 2500) {
|
|
149
|
+
const { model, host } = target;
|
|
150
|
+
const started = Date.now();
|
|
151
|
+
const out = await post(host, '/api/chat', {
|
|
152
|
+
model,
|
|
153
|
+
messages: [
|
|
154
|
+
{ role: 'system', content: readBrief() },
|
|
155
|
+
{ role: 'user', content: promptFor(situation) },
|
|
156
|
+
],
|
|
157
|
+
stream: false,
|
|
158
|
+
format: SCHEMA,
|
|
159
|
+
// Qwen3 reasons out loud by default. Turned off for two reasons and both
|
|
160
|
+
// are about fairness rather than speed: the Apple arm does not deliberate
|
|
161
|
+
// either, and a supervisor that takes twenty seconds to answer has already
|
|
162
|
+
// lost the argument it is here to have — it sits in front of a 1.2s budget.
|
|
163
|
+
think: false,
|
|
164
|
+
options: { temperature: 0, num_predict: 32 },
|
|
165
|
+
}, timeoutMs);
|
|
166
|
+
if (!out.json) return { kind: out.kind, error: out.error };
|
|
167
|
+
const ms = Date.now() - started;
|
|
168
|
+
try {
|
|
169
|
+
return { ...JSON.parse(out.json.message?.content ?? ''), ms };
|
|
170
|
+
} catch {
|
|
171
|
+
return { kind: 'unparseable', error: clip(out.json.message?.content, 200), ms };
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* Load the weights before timing anything.
|
|
177
|
+
*
|
|
178
|
+
* The Apple arm calls `prewarm()` for exactly this reason, and it was worth
|
|
179
|
+
* ~280ms there. Here it is worth minutes: an 8B at 4-bit is 5.2 GB off disk,
|
|
180
|
+
* and the first `doctor` probe aborted at 20s and reported a working model as
|
|
181
|
+
* one that "did not answer" — a cold load reported as a fault. Ollama's
|
|
182
|
+
* documented preload is a chat with no messages.
|
|
183
|
+
*/
|
|
184
|
+
export async function preload(target, timeoutMs = 120_000) {
|
|
185
|
+
const { model, host } = target;
|
|
186
|
+
const started = Date.now();
|
|
187
|
+
const out = await post(host, '/api/chat', { model, messages: [], keep_alive: '10m' }, timeoutMs);
|
|
188
|
+
return out.json ? { ok: true, ms: Date.now() - started } : { ok: false, ...out };
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** For `doctor`: is this arm actually there, and does it answer? */
|
|
192
|
+
export async function status(target, timeoutMs = 20_000) {
|
|
193
|
+
const { model, host } = target;
|
|
194
|
+
const tags = await post(host, '/api/show', { model }, 4000);
|
|
195
|
+
if (!tags.json) {
|
|
196
|
+
return {
|
|
197
|
+
ok: false,
|
|
198
|
+
reason: tags.kind === 'unreachable'
|
|
199
|
+
? `no Ollama server at ${host} (start it, or pass a host with ollama:<model>@<url>)`
|
|
200
|
+
: `Ollama has no model "${model}" (${tags.error})`,
|
|
201
|
+
};
|
|
202
|
+
}
|
|
203
|
+
return { ok: true, model, host, timeoutMs };
|
|
204
|
+
}
|
package/src/platform/ios.js
CHANGED
|
@@ -66,6 +66,19 @@ async function bootedDevices(opts) {
|
|
|
66
66
|
*/
|
|
67
67
|
async function resolveDevice(query, opts) {
|
|
68
68
|
const all = await listDevices(opts);
|
|
69
|
+
return pickDevice(query, all);
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Which device a query means, given the whole list.
|
|
74
|
+
*
|
|
75
|
+
* Separated from the listing so the decision can be **tested** rather than
|
|
76
|
+
* reasoned about, because two peer rounds in a row reported being handed the
|
|
77
|
+
* wrong device and both were decided here. The same move as `decisionOf` in the
|
|
78
|
+
* supervisor: the one line where a wrong answer is expensive should be a
|
|
79
|
+
* function somebody can call with adversarial input.
|
|
80
|
+
*/
|
|
81
|
+
export function pickDevice(query, all) {
|
|
69
82
|
const booted = all.filter((d) => d.state === 'Booted');
|
|
70
83
|
if (!query) {
|
|
71
84
|
if (booted.length === 0) throw new Error('no booted simulator (open Simulator.app or run `xcrun simctl boot <udid>`)');
|
|
@@ -98,8 +111,40 @@ async function resolveDevice(query, opts) {
|
|
|
98
111
|
const q = query.toLowerCase();
|
|
99
112
|
const pools = [booted, all];
|
|
100
113
|
for (const pool of pools) {
|
|
101
|
-
|
|
102
|
-
|
|
114
|
+
// A UDID is unique, so an exact UDID match needs no further thought.
|
|
115
|
+
const byUdid = pool.find((d) => d.udid.toLowerCase() === q);
|
|
116
|
+
if (byUdid) return byUdid;
|
|
117
|
+
// A **name is not unique**, and this branch used to treat it as though it
|
|
118
|
+
// were: one `find` over both fields returned whichever device the list
|
|
119
|
+
// happened to put first, and short-circuited past the ambiguity guard
|
|
120
|
+
// below. Item 83 recorded that two booted devices on this machine are both
|
|
121
|
+
// called "iPhone 17 Pro" and answered it by warning in `sim_devices` and
|
|
122
|
+
// printing a UDID prefix in headers — leaving the resolver, which is where
|
|
123
|
+
// the choice is actually made, untouched.
|
|
124
|
+
//
|
|
125
|
+
// What that cost, reported from a three-hour session on a real app: a
|
|
126
|
+
// caller passed the shared name, read a screen that was "38ms old" and an
|
|
127
|
+
// hour wrong, concluded the app had signed itself out, and abandoned a
|
|
128
|
+
// verification run that was fine. The frame was fresh — it was the *other*
|
|
129
|
+
// device's, idling on a login screen. `refresh: true` returned the matching
|
|
130
|
+
// tree because it refreshed the same wrong device. Two independent-looking
|
|
131
|
+
// sources agreeing with each other and both wrong.
|
|
132
|
+
//
|
|
133
|
+
// So a name that names two devices is an ambiguity, exactly like a partial
|
|
134
|
+
// match that hits two, and it refuses for the same reason: the cost of
|
|
135
|
+
// guessing wrong is reading somebody else's screen and believing it.
|
|
136
|
+
const byName = pool.filter((d) => d.name.toLowerCase() === q);
|
|
137
|
+
if (byName.length === 1) return byName[0];
|
|
138
|
+
if (byName.length > 1) {
|
|
139
|
+
throw Object.assign(
|
|
140
|
+
new Error(
|
|
141
|
+
`"${query}" is the name of ${byName.length} devices: `
|
|
142
|
+
+ `${byName.map((d) => d.udid).join(', ')} — a name cannot say which one you mean, `
|
|
143
|
+
+ 'so pass the UDID',
|
|
144
|
+
),
|
|
145
|
+
{ ambiguous: true },
|
|
146
|
+
);
|
|
147
|
+
}
|
|
103
148
|
const partial = pool.filter((d) => d.name.toLowerCase().includes(q));
|
|
104
149
|
if (partial.length === 1) return partial[0];
|
|
105
150
|
if (partial.length > 1) {
|
package/src/screenmap.js
CHANGED
|
@@ -151,6 +151,82 @@ const inside = (point, frame) =>
|
|
|
151
151
|
point.y >= frame.y &&
|
|
152
152
|
point.y <= frame.y + frame.height;
|
|
153
153
|
|
|
154
|
+
const frameArea = (f) => (f ? Math.max(1, f.width) * Math.max(1, f.height) : Infinity);
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* What the map says is at a point — everything containing it, smallest first.
|
|
158
|
+
*
|
|
159
|
+
* Item 120. Six consecutive `swipe [201,750] -> [201,250]` reported
|
|
160
|
+
* `no visible change` while a support banner sat at y≈753 swallowing every
|
|
161
|
+
* gesture. The banner was *in the element list the same call printed*; nothing
|
|
162
|
+
* connected "your swipe started at y=750" to "there is an element at y=753".
|
|
163
|
+
* The geometry was already in hand, so this is arithmetic over data we hold,
|
|
164
|
+
* not a new perception pass.
|
|
165
|
+
*
|
|
166
|
+
* Smallest first is a deliberately weaker claim than z-order. We do not know
|
|
167
|
+
* what is on top — the accessibility tree's order is not a paint order and OCR
|
|
168
|
+
* has none at all — and the honest statement is "these are the elements that
|
|
169
|
+
* cover that point", innermost first because the innermost is the one a
|
|
170
|
+
* gesture most often goes to. Saying "overlay" would be a guess wearing the
|
|
171
|
+
* clothes of a measurement.
|
|
172
|
+
*/
|
|
173
|
+
export function hitTest(entry, point) {
|
|
174
|
+
if (!entry?.targets || !Number.isFinite(point?.x) || !Number.isFinite(point?.y)) return [];
|
|
175
|
+
return entry.targets
|
|
176
|
+
.filter((t) => t.frame && inside(point, t.frame))
|
|
177
|
+
.sort((a, b) => frameArea(a.frame) - frameArea(b.frame));
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
const describeTarget = (t) => {
|
|
181
|
+
const name = t.label || t.text || (t.type ? `(unlabelled ${t.type})` : '(unlabelled)');
|
|
182
|
+
return t.region ? `"${name}" (${t.region})` : `"${name}"`;
|
|
183
|
+
};
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* One line saying what a coordinate resolved to, for a gesture that did
|
|
187
|
+
* nothing visible.
|
|
188
|
+
*
|
|
189
|
+
* Returns `null` when there is no map for the screen, because "we did not
|
|
190
|
+
* look" and "we looked and found nothing" are different answers and only one
|
|
191
|
+
* of them is worth printing.
|
|
192
|
+
*/
|
|
193
|
+
export function describePoint(entry, point, { what = 'the point' } = {}) {
|
|
194
|
+
if (!entry?.targets?.length) return null;
|
|
195
|
+
const at = `${Math.round(point.x)},${Math.round(point.y)}`;
|
|
196
|
+
const hits = hitTest(entry, point);
|
|
197
|
+
if (!hits.length) {
|
|
198
|
+
return `${what} ${at} is not inside any element on the map`
|
|
199
|
+
+ ' — empty space, or a view with no label (nothing can be said about what caught it)';
|
|
200
|
+
}
|
|
201
|
+
// The contention clause only where there is contention. On one hit the
|
|
202
|
+
// element's name *is* the diagnosis and anything after it is noise — and a
|
|
203
|
+
// note that pads every case is how a real one stops being read.
|
|
204
|
+
const others = hits.length > 1
|
|
205
|
+
? `, the smallest of ${hits.length} elements covering it — a gesture goes to whatever is on top there`
|
|
206
|
+
: '';
|
|
207
|
+
return `${what} ${at} is inside ${describeTarget(hits[0])}${others}`;
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
/**
|
|
211
|
+
* The point a step is aimed at, when it is aimed at a coordinate at all.
|
|
212
|
+
*
|
|
213
|
+
* A swipe is captured by whatever sits under where the finger goes *down*, so
|
|
214
|
+
* the start point is the one worth diagnosing; the end point never decides who
|
|
215
|
+
* receives the gesture.
|
|
216
|
+
*/
|
|
217
|
+
export function aimedAt(step) {
|
|
218
|
+
if (!step) return null;
|
|
219
|
+
if (step.action === 'tapAt' && Number.isFinite(step.x) && Number.isFinite(step.y)) {
|
|
220
|
+
return { point: { x: step.x, y: step.y }, what: 'the tap point' };
|
|
221
|
+
}
|
|
222
|
+
if (step.action === 'swipe') {
|
|
223
|
+
const x = step.from?.[0] ?? step.from?.x;
|
|
224
|
+
const y = step.from?.[1] ?? step.from?.y;
|
|
225
|
+
if (Number.isFinite(x) && Number.isFinite(y)) return { point: { x, y }, what: 'the swipe start point' };
|
|
226
|
+
}
|
|
227
|
+
return null;
|
|
228
|
+
}
|
|
229
|
+
|
|
154
230
|
/**
|
|
155
231
|
* Build the map for the screen currently showing. Accessibility elements are the
|
|
156
232
|
* real hit targets, so they win where they exist; OCR fills in everything the
|
package/src/supervisor.js
CHANGED
|
@@ -33,6 +33,7 @@
|
|
|
33
33
|
import path from 'node:path';
|
|
34
34
|
import { fileURLToPath } from 'node:url';
|
|
35
35
|
import * as store from './store.js';
|
|
36
|
+
import * as ollama from './ollama.js';
|
|
36
37
|
import { compiler, lineServer } from './localhelper.js';
|
|
37
38
|
|
|
38
39
|
const SOURCE = path.join(path.dirname(fileURLToPath(import.meta.url)), '..', 'native', 'supervise.swift');
|
|
@@ -45,6 +46,32 @@ const helper = lineServer({
|
|
|
45
46
|
|
|
46
47
|
export const DECISIONS = new Set(['wait', 'retry', 'stop']);
|
|
47
48
|
|
|
49
|
+
/**
|
|
50
|
+
* Which arm answers.
|
|
51
|
+
*
|
|
52
|
+
* Two, and the second one is an experiment rather than a recommendation. The
|
|
53
|
+
* owner asked for the capacity question to be settled with numbers instead of
|
|
54
|
+
* speculation, and a comparison needs something to compare against. `apple` is
|
|
55
|
+
* the shipped arm; `ollama:<model>` asks a local Ollama, off unless named.
|
|
56
|
+
*
|
|
57
|
+
* A backend is an object with `ask` and `status` and nothing else, which is the
|
|
58
|
+
* same shape the platform boundary uses one layer down — so `judge` below
|
|
59
|
+
* cannot tell which arm it asked, including in the failure shapes.
|
|
60
|
+
*/
|
|
61
|
+
function backendFor(want) {
|
|
62
|
+
if (want === 'apple') return { kind: 'apple', ask: helper.ask, status: appleStatus };
|
|
63
|
+
if (want === 'ollama' || want.startsWith('ollama:')) {
|
|
64
|
+
const target = ollama.parseTarget(want);
|
|
65
|
+
return {
|
|
66
|
+
kind: 'ollama',
|
|
67
|
+
target,
|
|
68
|
+
ask: (situation, timeoutMs) => ollama.ask(target, situation, timeoutMs),
|
|
69
|
+
status: () => ollamaStatus(target),
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
return null;
|
|
73
|
+
}
|
|
74
|
+
|
|
48
75
|
/**
|
|
49
76
|
* The vocabulary gate, as a function so it can be *tested* rather than grepped.
|
|
50
77
|
*
|
|
@@ -88,9 +115,15 @@ export async function judge({
|
|
|
88
115
|
goal, step, expected, failure, screen, stillMs, note, options, timeoutMs = 2500,
|
|
89
116
|
detail,
|
|
90
117
|
} = {}) {
|
|
91
|
-
|
|
118
|
+
const want = requested(options);
|
|
119
|
+
if (!want) return null;
|
|
92
120
|
if (!step || !failure) return null;
|
|
93
|
-
const
|
|
121
|
+
const backend = backendFor(want);
|
|
122
|
+
if (!backend) {
|
|
123
|
+
if (detail && typeof detail === 'object') detail.kind = `no such supervisor backend: "${want}"`;
|
|
124
|
+
return null;
|
|
125
|
+
}
|
|
126
|
+
const answer = await backend.ask({
|
|
94
127
|
goal: goal ? String(goal).slice(0, 200) : null,
|
|
95
128
|
step: String(step).slice(0, 200),
|
|
96
129
|
expected: expected ? String(expected).slice(0, 300) : null,
|
|
@@ -124,7 +157,59 @@ export async function judge({
|
|
|
124
157
|
export async function status(options) {
|
|
125
158
|
const want = requested(options);
|
|
126
159
|
if (!want) return { supervisor: 'none', detail: 'not requested (SIMFRAME_SUPERVISOR is unset)' };
|
|
127
|
-
|
|
160
|
+
const backend = backendFor(want);
|
|
161
|
+
if (!backend) return { supervisor: 'none', detail: `no such supervisor backend: "${want}"` };
|
|
162
|
+
return backend.status();
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* The experiment arm, reported exactly as honestly as the shipped one.
|
|
167
|
+
*
|
|
168
|
+
* Same probe, because "available" from a presence check is the failure this
|
|
169
|
+
* project has already paid for: the model had stopped answering inside the
|
|
170
|
+
* long-lived server while `doctor`, in its own process, said it was healthy —
|
|
171
|
+
* for twenty calls and six failures during which nothing was being judged.
|
|
172
|
+
*
|
|
173
|
+
* A wide probe budget on purpose. This arm loads several gigabytes on first
|
|
174
|
+
* ask, and a cold load timing out would report a working model as broken —
|
|
175
|
+
* which is a different lie from the one above and just as useless.
|
|
176
|
+
*/
|
|
177
|
+
async function ollamaStatus(target) {
|
|
178
|
+
const live = await ollama.status(target);
|
|
179
|
+
if (!live.ok) return { supervisor: 'none', detail: live.reason };
|
|
180
|
+
// Weights first, then the clock. Without this the probe measured a disk read
|
|
181
|
+
// and called a working model broken.
|
|
182
|
+
const warm = await ollama.preload(target);
|
|
183
|
+
if (!warm.ok) {
|
|
184
|
+
return {
|
|
185
|
+
supervisor: 'none',
|
|
186
|
+
detail: warm.kind === 'timeout'
|
|
187
|
+
? `"${target.model}" is still loading after 120s — ask again once it is resident`
|
|
188
|
+
: `"${target.model}" would not load (${warm.kind}: ${warm.error})`,
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
const probe = await ollama.ask(target, {
|
|
192
|
+
step: 'tap "Probe"',
|
|
193
|
+
failure: '"Probe" is not on this screen. Visible: Probe',
|
|
194
|
+
screen: ['Probe'],
|
|
195
|
+
stillMs: 5000,
|
|
196
|
+
}, live.timeoutMs);
|
|
197
|
+
if (decisionOf(probe) == null) {
|
|
198
|
+
return {
|
|
199
|
+
supervisor: 'none',
|
|
200
|
+
detail: `Ollama has "${target.model}" and it did not answer a probe`
|
|
201
|
+
+ `${probe?.kind ? ` (${probe.kind}${probe.error ? `: ${probe.error}` : ''})` : ''}`,
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
return {
|
|
205
|
+
supervisor: `ollama:${target.model}`,
|
|
206
|
+
detail: `${target.model} via Ollama at ${target.host}; answered a probe in ${probe.ms ?? '?'}ms;`
|
|
207
|
+
+ ' schema-constrained to wait/retry/stop. An experiment arm, not a recommendation —'
|
|
208
|
+
+ ' see docs/EXPERIMENTS.md for what it measured.',
|
|
209
|
+
};
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
async function appleStatus() {
|
|
128
213
|
const live = await helper.status();
|
|
129
214
|
if (!live.ok) return { supervisor: 'none', detail: live.reason };
|
|
130
215
|
// Prove a round trip, not a presence.
|