simframe 0.17.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -0
- package/package.json +1 -1
- package/scripts/bench-hpi.mjs +8 -0
- package/scripts/ci-device-guard.mjs +27 -0
- package/scripts/device-state.mjs +5 -62
- package/src/actions.js +18 -1
- package/src/cli.js +67 -4
- package/src/device-state.js +64 -0
- package/src/metrics.js +144 -3
- package/src/navigate.js +10 -0
- package/src/wedge.js +218 -0
package/README.md
CHANGED
|
@@ -751,6 +751,30 @@ project: **there is nothing left to win inside the engine.** 1.7 s is small
|
|
|
751
751
|
beside a 20 s round trip, so making perception or input faster buys single-digit
|
|
752
752
|
percentages. The only variable that matters is `n` — how many steps one decision
|
|
753
753
|
covers. Every improvement here has come from raising it, not from faster code.
|
|
754
|
+
|
|
755
|
+
### One more term: the launch is a fixed cost
|
|
756
|
+
|
|
757
|
+
Measured on the benchmark suite with **no model in the loop at all**:
|
|
758
|
+
|
|
759
|
+
| route | steps | agent | per step | human | per step |
|
|
760
|
+
| --- | --- | --- | --- | --- | --- |
|
|
761
|
+
| contacts-kate-bell | 2 | 12419 ms | **6.2 s** | 4300 ms | 2.15 s |
|
|
762
|
+
| settings-larger-text | 4 | 15281 ms | 3.8 s | 7799 ms | 1.95 s |
|
|
763
|
+
|
|
764
|
+
2.9× a human on the short route with nothing thinking, which does not fit
|
|
765
|
+
`1.7s × n`. A cold app launch is a **one-off cost amortised over the route**:
|
|
766
|
+
|
|
767
|
+
```
|
|
768
|
+
per step = (model round trip + launch cost + ~1.7s × n) / n
|
|
769
|
+
```
|
|
770
|
+
|
|
771
|
+
That reconciles a 7-step replay at 1.82 s/step with a 2-step route at 6.2 s/step
|
|
772
|
+
— one launch spread over 7 steps or over 2. **The `~1.7 s` figure is warm taps
|
|
773
|
+
inside a batch**, and short routes are materially worse than the table above
|
|
774
|
+
implies on its own.
|
|
775
|
+
|
|
776
|
+
`step_ratio` was **1** throughout: when a run completed it took exactly the
|
|
777
|
+
minimum number of steps.
|
|
754
778
|
A field report put the split at **34% simframe, 60% agent round trips** over
|
|
755
779
|
462 s of wall clock — the tester's *"30+ seconds between each step"* was
|
|
756
780
|
accurate and was not simframe. So there is nothing left to win inside the
|
|
@@ -871,6 +895,7 @@ simframe supervisions # local supervisor rulings, and what came of each
|
|
|
871
895
|
simframe hpi # speed and accuracy against a human baseline
|
|
872
896
|
simframe baseline record settings-larger-text --runs=5 # record the human
|
|
873
897
|
simframe input reset # rebuild the HID session, without restarting anything
|
|
898
|
+
simframe diagnose # what this device is doing right now, and which failure it is
|
|
874
899
|
simframe revive # power-cycle a wedged device: stop, shutdown, boot, start, reset input
|
|
875
900
|
simframe start / status / stop [--force] / devices
|
|
876
901
|
simframe ui --device=emulator-5554 # or export SIMFRAME_DEVICE once
|
|
@@ -895,6 +920,29 @@ rebinds, no frame since). This needs the device restarted —
|
|
|
895
920
|
`simframe revive --device=<udid>`. Backing off until a frame arrives.
|
|
896
921
|
```
|
|
897
922
|
|
|
923
|
+
**`simframe diagnose` says which failure it is**, which `doctor` cannot: `doctor`
|
|
924
|
+
answers "can this machine capture", and this answers "what is this device doing
|
|
925
|
+
right now". It reads only what discriminates — frame sequence, age and
|
|
926
|
+
stillness, element counts split by sensor, how many of them fuse, and who holds
|
|
927
|
+
the front by pid — and returns one of:
|
|
928
|
+
|
|
929
|
+
| verdict | what it means |
|
|
930
|
+
| --- | --- |
|
|
931
|
+
| `capture-down` | no frames at all, carrying the daemon's own sentence |
|
|
932
|
+
| `nothing-readable` | frames arriving, neither sensor finds a single element |
|
|
933
|
+
| `stale-frame` | both sensors full, almost nothing fuses — the framebuffer is behind the tree, so **an image from this device is not safe to trust** |
|
|
934
|
+
| `not-presenting` | an app holds the front by pid and the display shows almost nothing |
|
|
935
|
+
| `healthy` | — |
|
|
936
|
+
|
|
937
|
+
`not-presenting` deliberately does **not** say whether that is a lock screen, a
|
|
938
|
+
dead surface or a crashed system shell. It is not knowable from here, and
|
|
939
|
+
guessing is how a regex ended up standing where a measurement belongs.
|
|
940
|
+
|
|
941
|
+
The `stale-frame` threshold is measured rather than chosen: element fusion on
|
|
942
|
+
five healthy screens ran 0.667–0.929, so the threshold sits at 0.1 — 6.7× below
|
|
943
|
+
the observed floor rather than inside the metric's own noise. Numbers in
|
|
944
|
+
[`docs/BENCHMARKS.md`](docs/BENCHMARKS.md).
|
|
945
|
+
|
|
898
946
|
`simframe revive` is that restart, in the order that matters — stop the daemon,
|
|
899
947
|
shut the device down, boot it and *wait for the boot to finish*, start capture,
|
|
900
948
|
rebuild the HID session — and it ends by checking frames are flowing again
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "simframe",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.18.0",
|
|
4
4
|
"mcpName": "io.github.lvlrSajjad/simframe",
|
|
5
5
|
"description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
|
|
6
6
|
"keywords": [
|
package/scripts/bench-hpi.mjs
CHANGED
|
@@ -191,6 +191,14 @@ for (const f of report.flows) {
|
|
|
191
191
|
report.overall.hpi_time_median_of_passes = passTimes.length ? Number(metrics.median(passTimes).toFixed(3)) : null;
|
|
192
192
|
const o = report.overall;
|
|
193
193
|
console.log(`\nHPI_accuracy ${o.hpi_accuracy} HPI_time ${o.hpi_time ?? '—'} HPI ${o.hpi ?? '—'} step_ratio ${o.step_ratio ?? '—'}`);
|
|
194
|
+
if (o.runs_lost_to_device) {
|
|
195
|
+
// Measured on a full local suite: 5 of 17 runs failed because the guest's
|
|
196
|
+
// SpringBoard crashed or simctl stopped answering. They used to lower
|
|
197
|
+
// HPI_accuracy, which is the number this job gates on.
|
|
198
|
+
console.log(` over ${o.runs} measurable run(s); ${o.runs_lost_to_device} left the denominator`
|
|
199
|
+
+ ' because the DEVICE failed, not the code (item 173):');
|
|
200
|
+
for (const c of o.device_causes ?? []) console.log(` ${c}`);
|
|
201
|
+
}
|
|
194
202
|
if (passTimes.length > 1) {
|
|
195
203
|
console.log(`HPI_time per pass: ${passTimes.join(', ')} — median ${o.hpi_time_median_of_passes} (what the gate reads)`);
|
|
196
204
|
}
|
|
@@ -59,6 +59,25 @@ if (!cause) {
|
|
|
59
59
|
process.exit(first.code ?? 1);
|
|
60
60
|
}
|
|
61
61
|
|
|
62
|
+
// Diagnose BEFORE reviving, because reviving is what destroys the evidence.
|
|
63
|
+
//
|
|
64
|
+
// This guard has been recognising wedges from the shape of the failure text
|
|
65
|
+
// since 126, and then immediately power-cycling the device — so item 173 has
|
|
66
|
+
// accumulated a dozen occurrences and not one observation of what the device
|
|
67
|
+
// was doing at the time. `cause` above names a *consequence* ("a launched app
|
|
68
|
+
// never came to the front"); this names what the two sensors actually saw, and
|
|
69
|
+
// the difference decides between a stale framebuffer, a screen that is not the
|
|
70
|
+
// app, and capture being down. Its cost is one read on a path that is already
|
|
71
|
+
// failing.
|
|
72
|
+
const diagnose = async (when) => {
|
|
73
|
+
const d = await run(['node', 'src/cli.js', 'diagnose', `--device=${udid}`]);
|
|
74
|
+
console.error(`\n (diagnosis ${when} — for DEFERRED 173)`);
|
|
75
|
+
const verdict = /— (\S+)\n/.exec(d.out)?.[1] ?? 'unreadable';
|
|
76
|
+
summary(` - diagnosis ${when}: \`${verdict}\``);
|
|
77
|
+
return verdict;
|
|
78
|
+
};
|
|
79
|
+
const before = await diagnose('before the revive');
|
|
80
|
+
|
|
62
81
|
console.error(`\n (${cause} — DEFERRED 126. Reviving once and running again.)`);
|
|
63
82
|
await run(['node', 'src/cli.js', 'revive', `--device=${udid}`], { capture: false });
|
|
64
83
|
const second = await run(cmd);
|
|
@@ -70,4 +89,12 @@ if (second.code === 0) {
|
|
|
70
89
|
const again = deviceCause(second.out);
|
|
71
90
|
summary(`- \`${cmd.join(' ')}\` — **${again ? 'device unavailable' : 'check failed'}** after a revive${again ? ` (${again})` : ''}`);
|
|
72
91
|
if (again) console.error(`\nFAIL the simulator is still in a bad state after a revive: ${again}`);
|
|
92
|
+
// Twice is the interesting case: a revive cured it and it came back, or the
|
|
93
|
+
// revive did not cure it at all. Those are different faults and the pair of
|
|
94
|
+
// diagnoses says which.
|
|
95
|
+
const after = await diagnose('after the revive');
|
|
96
|
+
if (before !== after) {
|
|
97
|
+
console.error(`\n (the device changed state across the revive: ${before} -> ${after})`);
|
|
98
|
+
summary(` - state changed across the revive: \`${before}\` -> \`${after}\``);
|
|
99
|
+
}
|
|
73
100
|
process.exit(second.code ?? 1);
|
package/scripts/device-state.mjs
CHANGED
|
@@ -1,62 +1,5 @@
|
|
|
1
|
-
//
|
|
2
|
-
//
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
|
|
6
|
-
// and the only thing that exercised it was a hosted runner, at the end of a
|
|
7
|
-
// fifteen-minute job, in the middle of a report. Two runtime bugs in this
|
|
8
|
-
// project came from logic that was correct and had never executed.
|
|
9
|
-
//
|
|
10
|
-
// Every entry here has been seen in a real run. Adding one from imagination is
|
|
11
|
-
// how a guard starts reviving genuine failures into passes.
|
|
12
|
-
|
|
13
|
-
/** Conditions that are the simulator, not the code. Each seen in a real run. */
|
|
14
|
-
export const DEVICE_STATE = [
|
|
15
|
-
[/NSPOSIXErrorDomain.*code=?\s*60|Operation timed out/i, 'simctl stopped answering (NSPOSIXErrorDomain 60)'],
|
|
16
|
-
[/did not produce a frame|produced no frame in \d+s/i, 'the daemon is up and the display renders nothing'],
|
|
17
|
-
[/Timeout waiting for screen surfaces|display surface is not answering|display surface could not be read/i, 'the display surface is wedged'],
|
|
18
|
-
[/no frames buffered|capture is wedged/i, 'capture stopped'],
|
|
19
|
-
[/the second app never launched|could not be dispatched/i, 'an app would not launch'],
|
|
20
|
-
// A launched app that never comes to the front, seen as the tour waiting for
|
|
21
|
-
// one of its landmarks on a screen that is showing a clock and nothing else.
|
|
22
|
-
//
|
|
23
|
-
// Measured on a runner: `ok launch — launched com.apple.Preferences
|
|
24
|
-
// (relaunched)` followed by `waited 8000ms for General: "General" is not on
|
|
25
|
-
// this screen. Visible: 10:50, .?o (the screen has not moved for 6181ms)`.
|
|
26
|
-
// Two labels, one of them a clock, on a still screen — the device is not
|
|
27
|
-
// presenting the app, and the guard called that a check failing on its
|
|
28
|
-
// merits and declined to revive.
|
|
29
|
-
//
|
|
30
|
-
// Deliberately narrow. It requires the wait to have failed AND the screen to
|
|
31
|
-
// have been still AND almost nothing readable: a tour that genuinely asks for
|
|
32
|
-
// the wrong label has a screen full of other labels, and must keep failing
|
|
33
|
-
// rather than being retried into a pass.
|
|
34
|
-
[
|
|
35
|
-
/never arrived[\s\S]*?Visible:[^\n]{0,24}\(the screen has not moved for \d+ms/i,
|
|
36
|
-
'a launched app never came to the front (the screen shows a clock and nothing else)',
|
|
37
|
-
],
|
|
38
|
-
// The same condition, now said outright by the step that suffered it instead
|
|
39
|
-
// of inferred from the shape of the screen afterwards. Item 169 gave `launch`
|
|
40
|
-
// a pid to compare, so a launch that starts a process and never fronts it
|
|
41
|
-
// reports itself; this signature fires on the cause rather than on a
|
|
42
|
-
// consequence that had to be recognised by "two labels, one a clock".
|
|
43
|
-
//
|
|
44
|
-
// It cannot be triggered by a tour asking for the wrong label — only a failed
|
|
45
|
-
// launch emits this sentence — so it needs none of the narrowing above.
|
|
46
|
-
[
|
|
47
|
-
/never came to the front within \d+ms/i,
|
|
48
|
-
'a launched app never came to the front (the launch said so itself, by pid)',
|
|
49
|
-
],
|
|
50
|
-
// Seen on the v0.14.3 bench run: `could not launch com.apple.Preferences:
|
|
51
|
-
// The system shell (SpringBoard:36454) probably crashed.` The guest's window
|
|
52
|
-
// server going down is the device, not the check, and nothing here matched it.
|
|
53
|
-
[
|
|
54
|
-
/system shell \(SpringBoard[^)]*\) probably crashed/i,
|
|
55
|
-
"the guest's SpringBoard crashed, so nothing can be fronted",
|
|
56
|
-
],
|
|
57
|
-
];
|
|
58
|
-
|
|
59
|
-
/** The condition this output shows, or null when the check failed on its merits. */
|
|
60
|
-
export function deviceCause(text) {
|
|
61
|
-
return DEVICE_STATE.find(([re]) => re.test(String(text ?? '')))?.[1] ?? null;
|
|
62
|
-
}
|
|
1
|
+
// Kept as a re-export: the table moved to `src/device-state.js` so `src/` can
|
|
2
|
+
// use it without importing out of `scripts/`. `ci-device-guard.mjs` and the
|
|
3
|
+
// unit test both reach it through this path, and a redirect is cheaper than
|
|
4
|
+
// updating every caller for a move that changes nothing about the table.
|
|
5
|
+
export { DEVICE_STATE, deviceCause } from '../src/device-state.js';
|
package/src/actions.js
CHANGED
|
@@ -933,7 +933,19 @@ export async function runScript(
|
|
|
933
933
|
// switch moving 0.1% of the screen, which is neither faculty. Saying
|
|
934
934
|
// "assumed" is the honest answer; guessing a better-sounding reason
|
|
935
935
|
// would be the same mistake in the other direction.
|
|
936
|
-
|
|
936
|
+
//
|
|
937
|
+
// **Per verdict, though, not per site.** That argument is about
|
|
938
|
+
// `no-visible-change` and does not extend to every verdict: an
|
|
939
|
+
// `unexpected-screen` means the screen after the action was not the
|
|
940
|
+
// one memory predicted, which is item 174 and nothing else.
|
|
941
|
+
// `metrics.VERDICT_FACULTY` holds the verdicts whose faculty is read,
|
|
942
|
+
// and `no-visible-change` is deliberately not one of them — so this
|
|
943
|
+
// still says "assumed" for the 162 records the reasoning above is
|
|
944
|
+
// actually about, and stops saying it for the 26 it never covered.
|
|
945
|
+
classified: Boolean(metrics.VERDICT_FACULTY[verification?.verdict]),
|
|
946
|
+
// Recorded as a field rather than left as a prefix of `detail`, which
|
|
947
|
+
// is how the breakdown had to recover it: by parsing a string.
|
|
948
|
+
verdict: verification?.verdict ?? null,
|
|
937
949
|
// `verification_failed` is the largest reason class in the log and it
|
|
938
950
|
// was the only one carrying no intent, which made most of the corpus
|
|
939
951
|
// useless for asking what kind of decision costs us. The step knows
|
|
@@ -989,6 +1001,11 @@ export async function runScript(
|
|
|
989
1001
|
const why = metrics.reasonForStepError(step, err);
|
|
990
1002
|
noteEscalation({
|
|
991
1003
|
stepIndex: i,
|
|
1004
|
+
// When the failure is the simulator rather than the code, say so. 78 of
|
|
1005
|
+
// the bench device's `verification_failed` records are `simctl` failing,
|
|
1006
|
+
// an app that would not launch, or capture stopping — item 173, counted
|
|
1007
|
+
// towards a perception phase.
|
|
1008
|
+
device: why.device ?? null,
|
|
992
1009
|
fingerprint: beforeScreen?.hash ?? metrics.fingerprintNow(udid, screenmap),
|
|
993
1010
|
reason: why.reason,
|
|
994
1011
|
candidates: why.candidates,
|
package/src/cli.js
CHANGED
|
@@ -11,6 +11,7 @@ import * as input from './input.js';
|
|
|
11
11
|
import * as baseline from './baseline.js';
|
|
12
12
|
import * as metrics from './metrics.js';
|
|
13
13
|
import * as navigate from './navigate.js';
|
|
14
|
+
import * as wedge from './wedge.js';
|
|
14
15
|
import { decodePng } from './png.js';
|
|
15
16
|
import * as storage from './storage.js';
|
|
16
17
|
import * as store from './store.js';
|
|
@@ -51,6 +52,7 @@ const USAGE = `simframe — always-warm iOS Simulator frames
|
|
|
51
52
|
simframe escalations [device] why simframe handed decisions back, by reason
|
|
52
53
|
simframe supervisions [device] local supervisor rulings, and what came of each
|
|
53
54
|
simframe revive [device] power-cycle a wedged device: stop, shutdown, boot, start, reset input
|
|
55
|
+
simframe diagnose [device] what this device is doing right now, and which failure it is
|
|
54
56
|
(--session=<id> narrows to one agent; the
|
|
55
57
|
ids are listed in the output. SIMFRAME_SESSION
|
|
56
58
|
names one, but only at process start — an
|
|
@@ -447,6 +449,26 @@ async function main() {
|
|
|
447
449
|
return;
|
|
448
450
|
}
|
|
449
451
|
|
|
452
|
+
// Not folded into `doctor`, which answers "can this machine capture". This
|
|
453
|
+
// answers "what is this device doing right now", which is item 173's
|
|
454
|
+
// question and has never had an instrument.
|
|
455
|
+
case 'diagnose': {
|
|
456
|
+
const dev = await resolveDevice(device);
|
|
457
|
+
const r = await wedge.diagnose(dev.udid, { options });
|
|
458
|
+
emit(flags, r, [
|
|
459
|
+
`${r.device.name} — ${r.verdict.state}`,
|
|
460
|
+
` ${r.verdict.detail}`,
|
|
461
|
+
'',
|
|
462
|
+
` frame seq ${r.frame?.seq ?? '-'}, ${r.frame?.ageMs ?? '-'}ms old, still for ${r.frame?.stableForMs ?? '-'}ms, ${r.frame?.size ?? '-'}`,
|
|
463
|
+
` elements ${r.elements.total} total — ${r.elements.ax} by tree, ${r.elements.ocr} by OCR, ${r.elements.fused} by both`,
|
|
464
|
+
` fusion ${r.agreement ?? 'n/a'} of elements seen by both sensors (measured healthy ${wedge.HEALTHY_FUSION}; at or below ${wedge.DISAGREEMENT} the frame is stale)`,
|
|
465
|
+
` frontmost ${r.frontmost?.pid ?? 'unknown'}${r.frontmost?.title ? ` (${r.frontmost.title})` : ''}`,
|
|
466
|
+
...(r.verdict.revive ? ['', ' `simframe revive` is the recovery. Keep this output — item 173 needs it.'] : []),
|
|
467
|
+
]);
|
|
468
|
+
if (r.verdict.revive) process.exitCode = 1;
|
|
469
|
+
return;
|
|
470
|
+
}
|
|
471
|
+
|
|
450
472
|
case 'status': {
|
|
451
473
|
const udids = device
|
|
452
474
|
? [(await resolveDevice(device)).udid]
|
|
@@ -1253,8 +1275,14 @@ async function main() {
|
|
|
1253
1275
|
'flow runs agent p50 human p50 HPI_time step_ratio turns esc',
|
|
1254
1276
|
...report.flows.map((f) => metrics.flowRow(f, { wide: true })),
|
|
1255
1277
|
'',
|
|
1256
|
-
`HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs}
|
|
1278
|
+
`HPI_accuracy ${report.overall.hpi_accuracy} (${report.overall.runs} measurable run(s), ` +
|
|
1257
1279
|
`${report.overall.runs - runs.filter((r) => r.completed && !r.wrong_action_taken).length} not clean)`,
|
|
1280
|
+
// A denominator that quietly shrinks is worse than one that is wrong.
|
|
1281
|
+
report.overall.runs_lost_to_device
|
|
1282
|
+
? ` ${report.overall.runs_lost_to_device} further run(s) left the denominator because the DEVICE failed,`
|
|
1283
|
+
+ ' not the code — those are unmeasured, not inaccurate (item 173):'
|
|
1284
|
+
+ `\n${(report.overall.device_causes ?? []).map((c) => ` ${c}`).join('\n')}`
|
|
1285
|
+
: null,
|
|
1258
1286
|
report.overall.hpi_time == null
|
|
1259
1287
|
? `HPI_time and HPI need a human baseline — none of ${report.overall.flows_measured} measured flow(s) has one yet.`
|
|
1260
1288
|
: `HPI_time ${report.overall.hpi_time} (harmonic mean over ${report.overall.flows_with_human_baseline} flow(s)), HPI ${report.overall.hpi}`,
|
|
@@ -1265,7 +1293,10 @@ async function main() {
|
|
|
1265
1293
|
`steps per model call ${report.overall.steps_per_call ?? '—'}`
|
|
1266
1294
|
+ (report.overall.steps_per_call
|
|
1267
1295
|
? ` — about ${(((20000 + 1700 * report.overall.steps_per_call) / report.overall.steps_per_call) / 1000).toFixed(1)}s`
|
|
1268
|
-
+ ' per step end to end
|
|
1296
|
+
+ ' per step end to end. Raise this, not the engine — but note the ~1.7s'
|
|
1297
|
+
+ ' figure for simframe\'s own work is WARM taps inside a batch: a cold app'
|
|
1298
|
+
+ ' launch is a large one-off on top, and it dominates short flows'
|
|
1299
|
+
+ ' (measured: 6.2s/step over 2 steps with no model at all).'
|
|
1269
1300
|
: ''),
|
|
1270
1301
|
flags.out ? `wrote ${flags.out}` : null,
|
|
1271
1302
|
]);
|
|
@@ -1354,13 +1385,45 @@ async function main() {
|
|
|
1354
1385
|
if (read > 0) {
|
|
1355
1386
|
verdict = named + (read < n ? ` (on the ${read} of ${n} whose reason was read)` : '');
|
|
1356
1387
|
} else if (assumed > 0) {
|
|
1357
|
-
|
|
1388
|
+
// For `verification_failed` this line used to end the story, and
|
|
1389
|
+
// it is no longer the whole truth: the verdict split below names
|
|
1390
|
+
// a faculty for some of them and takes the device's own failures
|
|
1391
|
+
// out altogether.
|
|
1392
|
+
verdict = r === 'verification_failed' && (b.verdicts?.length || b.device_total)
|
|
1393
|
+
? 'reason too coarse to steer by — see the verdict split below'
|
|
1394
|
+
: 'reason assumed, not read — no faculty can be named from these';
|
|
1358
1395
|
} else {
|
|
1359
1396
|
// Neither read nor assumed: the log predates the distinction.
|
|
1360
|
-
verdict = `${named} — but these records predate the check, so treat it as untested
|
|
1397
|
+
verdict = `${named} — but these records predate the check, so treat it as untested`
|
|
1398
|
+
+ ' (legacy, not assumed)';
|
|
1361
1399
|
}
|
|
1362
1400
|
return ` ${r.padEnd(20)} ${String(n).padStart(4)} ${verdict}`;
|
|
1363
1401
|
}),
|
|
1402
|
+
// The two things the reason alone could not say.
|
|
1403
|
+
//
|
|
1404
|
+
// Both exist because the per-reason table above was the steering wheel
|
|
1405
|
+
// and it pointed at one faculty for a class holding three, and counted
|
|
1406
|
+
// the simulator's own failures towards a perception phase.
|
|
1407
|
+
b.verdicts?.length ? '' : null,
|
|
1408
|
+
b.verdicts?.length
|
|
1409
|
+
? 'verification_failed, split by the verdict that fired'
|
|
1410
|
+
+ (b.derived?.verdicts
|
|
1411
|
+
? ` (${b.derived.verdicts} of ${b.verdicts.reduce((a, v) => a + v.count, 0)} derived from the detail text, not recorded at the time):`
|
|
1412
|
+
: ':')
|
|
1413
|
+
: null,
|
|
1414
|
+
...(b.verdicts ?? []).slice(0, 8).map((v) => {
|
|
1415
|
+
const faculty = metrics.VERDICT_FACULTY[v.name];
|
|
1416
|
+
return ` ${v.name.padEnd(20)} ${String(v.count).padStart(4)} `
|
|
1417
|
+
+ (faculty
|
|
1418
|
+
? `points at: ${faculty}`
|
|
1419
|
+
: 'no faculty follows from this verdict alone — see metrics.VERDICT_FACULTY');
|
|
1420
|
+
}),
|
|
1421
|
+
b.device_total ? '' : null,
|
|
1422
|
+
b.device_total
|
|
1423
|
+
? `${b.device_total} escalation(s) were the DEVICE, not the code — item 173, and not evidence for any faculty`
|
|
1424
|
+
+ `${b.derived?.device ? ` (${b.derived.device} derived from the detail text)` : ''}:`
|
|
1425
|
+
: null,
|
|
1426
|
+
...(b.device ?? []).slice(0, 6).map((d) => ` ${String(d.count).padStart(4)} ${d.name}`),
|
|
1364
1427
|
b.total ? '' : null,
|
|
1365
1428
|
b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
|
|
1366
1429
|
// Said out loud rather than left for someone to discover: the rate is
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
// Is a failure the device's fault or the code's?
|
|
2
|
+
//
|
|
3
|
+
// Moved out of `scripts/` on 2026-09-17 so `src/metrics.js` can use it. The
|
|
4
|
+
// escalation log needed it: 78 of the `verification_failed` records on the
|
|
5
|
+
// bench device are `xcrun simctl openurl` failing, an app that would not
|
|
6
|
+
// launch, or capture stopping — the device, filed under a code faculty and
|
|
7
|
+
// reported as evidence for "sense of time (Phase 11)". A steering wheel that
|
|
8
|
+
// counts item 173's own occurrences as a perception problem points somewhere
|
|
9
|
+
// nobody chose.
|
|
10
|
+
//
|
|
11
|
+
// Deliberately dependency-free. `wedge.js` imports `index.js`, and `metrics.js`
|
|
12
|
+
// is imported *by* `index.js`, so a shared table living in either would close a
|
|
13
|
+
// cycle. This module imports nothing and is imported by both.
|
|
14
|
+
|
|
15
|
+
/** Conditions that are the simulator, not the code. Each seen in a real run. */
|
|
16
|
+
export const DEVICE_STATE = [
|
|
17
|
+
[/NSPOSIXErrorDomain.*code=?\s*60|Operation timed out/i, 'simctl stopped answering (NSPOSIXErrorDomain 60)'],
|
|
18
|
+
[/did not produce a frame|produced no frame in \d+s/i, 'the daemon is up and the display renders nothing'],
|
|
19
|
+
[/Timeout waiting for screen surfaces|display surface is not answering|display surface could not be read/i, 'the display surface is wedged'],
|
|
20
|
+
[/no frames buffered|capture is wedged/i, 'capture stopped'],
|
|
21
|
+
[/the second app never launched|could not be dispatched/i, 'an app would not launch'],
|
|
22
|
+
// A launched app that never comes to the front, seen as the tour waiting for
|
|
23
|
+
// one of its landmarks on a screen that is showing a clock and nothing else.
|
|
24
|
+
//
|
|
25
|
+
// Measured on a runner: `ok launch — launched com.apple.Preferences
|
|
26
|
+
// (relaunched)` followed by `waited 8000ms for General: "General" is not on
|
|
27
|
+
// this screen. Visible: 10:50, .?o (the screen has not moved for 6181ms)`.
|
|
28
|
+
// Two labels, one of them a clock, on a still screen — the device is not
|
|
29
|
+
// presenting the app, and the guard called that a check failing on its
|
|
30
|
+
// merits and declined to revive.
|
|
31
|
+
//
|
|
32
|
+
// Deliberately narrow. It requires the wait to have failed AND the screen to
|
|
33
|
+
// have been still AND almost nothing readable: a tour that genuinely asks for
|
|
34
|
+
// the wrong label has a screen full of other labels, and must keep failing
|
|
35
|
+
// rather than being retried into a pass.
|
|
36
|
+
[
|
|
37
|
+
/never arrived[\s\S]*?Visible:[^\n]{0,24}\(the screen has not moved for \d+ms/i,
|
|
38
|
+
'a launched app never came to the front (the screen shows a clock and nothing else)',
|
|
39
|
+
],
|
|
40
|
+
// The same condition, now said outright by the step that suffered it instead
|
|
41
|
+
// of inferred from the shape of the screen afterwards. Item 169 gave `launch`
|
|
42
|
+
// a pid to compare, so a launch that starts a process and never fronts it
|
|
43
|
+
// reports itself; this signature fires on the cause rather than on a
|
|
44
|
+
// consequence that had to be recognised by "two labels, one a clock".
|
|
45
|
+
//
|
|
46
|
+
// It cannot be triggered by a tour asking for the wrong label — only a failed
|
|
47
|
+
// launch emits this sentence — so it needs none of the narrowing above.
|
|
48
|
+
[
|
|
49
|
+
/never came to the front within \d+ms/i,
|
|
50
|
+
'a launched app never came to the front (the launch said so itself, by pid)',
|
|
51
|
+
],
|
|
52
|
+
// Seen on the v0.14.3 bench run: `could not launch com.apple.Preferences:
|
|
53
|
+
// The system shell (SpringBoard:36454) probably crashed.` The guest's window
|
|
54
|
+
// server going down is the device, not the check, and nothing here matched it.
|
|
55
|
+
[
|
|
56
|
+
/system shell \(SpringBoard[^)]*\) probably crashed/i,
|
|
57
|
+
"the guest's SpringBoard crashed, so nothing can be fronted",
|
|
58
|
+
],
|
|
59
|
+
];
|
|
60
|
+
|
|
61
|
+
/** The condition this output shows, or null when the check failed on its merits. */
|
|
62
|
+
export function deviceCause(text) {
|
|
63
|
+
return DEVICE_STATE.find(([re]) => re.test(String(text ?? '')))?.[1] ?? null;
|
|
64
|
+
}
|
package/src/metrics.js
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
import fs from 'node:fs';
|
|
13
13
|
import path from 'node:path';
|
|
14
14
|
import * as store from './store.js';
|
|
15
|
+
import { deviceCause } from './device-state.js';
|
|
15
16
|
|
|
16
17
|
/** The five reasons, from docs/research/03-human-parity.md §8. Nothing else is a reason. */
|
|
17
18
|
export const REASONS = [
|
|
@@ -66,6 +67,52 @@ export const FACULTY = {
|
|
|
66
67
|
*/
|
|
67
68
|
export const BUILT_FACULTIES = new Set(['sense of time (Phase 11)']);
|
|
68
69
|
|
|
70
|
+
/**
|
|
71
|
+
* The faculty a `verification_failed` record points at, by **verdict**.
|
|
72
|
+
*
|
|
73
|
+
* `FACULTY` is keyed on the reason, and for this class the reason is too coarse
|
|
74
|
+
* to steer by. Measured on the bench device's 1022 records: of the
|
|
75
|
+
* `verification_failed` ones, 162 are `no-visible-change`, 26 are
|
|
76
|
+
* `unexpected-screen`, and about 172 are a wait that timed out — three
|
|
77
|
+
* different faculties, all of which the report named as "sense of time
|
|
78
|
+
* (Phase 11)" because that is what the reason maps to.
|
|
79
|
+
*
|
|
80
|
+
* `unexpected-screen` is the clearest case: it is item 174, screen identity
|
|
81
|
+
* fragmenting on content-driven screens, and reporting it as a timing problem
|
|
82
|
+
* is how a tester was once told their unlabeled-control problem was a timing
|
|
83
|
+
* problem.
|
|
84
|
+
*/
|
|
85
|
+
/**
|
|
86
|
+
* The verdict a legacy record carries in the first token of its `detail`.
|
|
87
|
+
*
|
|
88
|
+
* Written as `${verdict}: ${detail}` since the field existed, so the prefix is
|
|
89
|
+
* reliable — but only for records whose detail came from a verdict at all,
|
|
90
|
+
* which is why this returns null rather than guessing on anything else.
|
|
91
|
+
*/
|
|
92
|
+
export function verdictFromDetail(detail) {
|
|
93
|
+
const m = /^([a-z][a-z-]{3,30}):\s/.exec(String(detail ?? ''));
|
|
94
|
+
// Only a verdict that exists. The first draft returned any lowercase prefix
|
|
95
|
+
// and duly reported a verdict called **"capture"** with a count of 8, from
|
|
96
|
+
// details reading `capture: ...`. A parser that invents a category gets it
|
|
97
|
+
// counted, named in a report, and eventually used to choose a phase — which
|
|
98
|
+
// is the whole failure this grouping was added to fix, reproduced inside the
|
|
99
|
+
// fix.
|
|
100
|
+
return m && KNOWN_VERDICTS.has(m[1]) ? m[1] : null;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export const VERDICT_FACULTY = {
|
|
104
|
+
'unexpected-screen': 'screen identity (item 174)',
|
|
105
|
+
'still-filling-in': 'sense of time (Phase 11)',
|
|
106
|
+
// **`no-visible-change` is deliberately absent, and it is the largest verdict
|
|
107
|
+
// in the log (162 of 1022).** The escalation site has a measured argument for
|
|
108
|
+
// leaving it unclassified and it is right: in the field it came
|
|
109
|
+
// overwhelmingly from tapping an inert text label whose real hit target was
|
|
110
|
+
// an invisible chevron — icon semantics, Phase 15 — but it also covers a
|
|
111
|
+
// switch moving 0.1% of the screen, which is neither faculty. Two causes, one
|
|
112
|
+
// verdict, and no way to tell them apart from here. Putting it in this map
|
|
113
|
+
// would name a faculty for 162 records on a coin flip.
|
|
114
|
+
};
|
|
115
|
+
|
|
69
116
|
function metricPaths(udid) {
|
|
70
117
|
const dir = store.deviceDir(udid);
|
|
71
118
|
return {
|
|
@@ -323,6 +370,17 @@ export function reasonForStepError(step, err) {
|
|
|
323
370
|
// admits "other" collects a pile of "other". What changes is that the record
|
|
324
371
|
// carries whether the reason was *read off the failure* or *assumed*, and
|
|
325
372
|
// the report declines to recommend a faculty for the assumed ones.
|
|
373
|
+
// The device, not a faculty.
|
|
374
|
+
//
|
|
375
|
+
// 78 of this class on the bench device are `xcrun simctl openurl` failing, an
|
|
376
|
+
// app that would not launch, or capture stopping. Those are item 173, and
|
|
377
|
+
// filing them under a code faculty is how the log came to offer "sense of
|
|
378
|
+
// time (Phase 11)" as the remedy for a simulator that had stopped answering.
|
|
379
|
+
// `classified` stays false because no *faculty* was read; `device` says what
|
|
380
|
+
// was, so the report can take these out of the faculty count instead of
|
|
381
|
+
// counting them towards a phase.
|
|
382
|
+
const device = deviceCause(err?.message);
|
|
383
|
+
if (device) return { reason: 'verification_failed', candidates: [], tried: [], classified: false, device };
|
|
326
384
|
return { reason: 'verification_failed', candidates: [], tried: [], classified: false };
|
|
327
385
|
}
|
|
328
386
|
|
|
@@ -336,6 +394,17 @@ export function reasonForStepError(step, err) {
|
|
|
336
394
|
*/
|
|
337
395
|
export const ESCALATING_VERDICTS = new Set(['unexpected-screen', 'no-visible-change']);
|
|
338
396
|
|
|
397
|
+
/**
|
|
398
|
+
* Every verdict the engine can report, for reading legacy details back.
|
|
399
|
+
*
|
|
400
|
+
* Wider than `ESCALATING_VERDICTS` — which is the two that hand back to a model
|
|
401
|
+
* — because a record's detail may carry any of them, and narrower than "any
|
|
402
|
+
* lowercase word", which is what a prefix parser accepts if nobody bounds it.
|
|
403
|
+
*/
|
|
404
|
+
export const KNOWN_VERDICTS = new Set([
|
|
405
|
+
'unexpected-screen', 'no-visible-change', 'still-filling-in', 'unverified', 'ok',
|
|
406
|
+
]);
|
|
407
|
+
|
|
339
408
|
/** How `goto`/`flow run` refusals map. They refuse rather than guess, and the refusal is the hand-back. */
|
|
340
409
|
export const PLAN_REASONS = {
|
|
341
410
|
'unknown-screen': 'unknown_screen',
|
|
@@ -463,6 +532,14 @@ export function recordEscalation(udid, {
|
|
|
463
532
|
// Default `false`, so a caller that does not think about it cannot
|
|
464
533
|
// accidentally claim precision it does not have.
|
|
465
534
|
classified = false,
|
|
535
|
+
// Which device-state condition this failure shows, when it shows one. Kept
|
|
536
|
+
// separate from `reason` because the five-reason vocabulary is fixed and a
|
|
537
|
+
// sixth reason collects a pile of "other" — see REASONS.
|
|
538
|
+
device = null,
|
|
539
|
+
// Which local verdict fired, for the records that have one. Derivable from
|
|
540
|
+
// `detail` today by parsing a prefix, which is exactly the fragility the log
|
|
541
|
+
// should not depend on.
|
|
542
|
+
verdict = null,
|
|
466
543
|
} = {}) {
|
|
467
544
|
if (!REASONS.includes(reason)) throw new Error(`not an escalation reason: ${reason}`);
|
|
468
545
|
if (!OUTCOMES.includes(outcome)) throw new Error(`not an escalation outcome: ${outcome}`);
|
|
@@ -479,6 +556,10 @@ export function recordEscalation(udid, {
|
|
|
479
556
|
// go/no-go, and on its own it answers "what kind of decision is costing us".
|
|
480
557
|
intent: intent ? String(intent).slice(0, 120) : null,
|
|
481
558
|
classified: Boolean(classified),
|
|
559
|
+
// Null unless read. Both of these separate "we know" from "we assumed" for
|
|
560
|
+
// a class that is otherwise one coarse bucket.
|
|
561
|
+
device_cause: device ? String(device).slice(0, 120) : null,
|
|
562
|
+
verdict: verdict ? String(verdict).slice(0, 60) : null,
|
|
482
563
|
step_index: stepIndex,
|
|
483
564
|
screen_fingerprint: fingerprint,
|
|
484
565
|
reason,
|
|
@@ -538,7 +619,9 @@ export function flowRecordFrom({
|
|
|
538
619
|
images_sent: imagesSent,
|
|
539
620
|
input_tokens: null,
|
|
540
621
|
output_tokens: null,
|
|
541
|
-
escalations: escalations.map((e) => ({
|
|
622
|
+
escalations: escalations.map((e) => ({
|
|
623
|
+
reason: e.reason, step_index: e.step_index, outcome: e.outcome, device_cause: e.device_cause ?? null,
|
|
624
|
+
})),
|
|
542
625
|
escalation_count: escalations.length,
|
|
543
626
|
mis_taps: misTaps,
|
|
544
627
|
verdict_histogram: histogram,
|
|
@@ -546,6 +629,22 @@ export function flowRecordFrom({
|
|
|
546
629
|
exploration_events: [],
|
|
547
630
|
completed: Boolean(completed),
|
|
548
631
|
wrong_action_taken: verdicts.includes('unexpected-screen'),
|
|
632
|
+
// Did this run fail because the *simulator* failed?
|
|
633
|
+
//
|
|
634
|
+
// Measured, on a full local suite: of 17 runs, 5 failed because the guest's
|
|
635
|
+
// SpringBoard crashed or `simctl` stopped answering for 90 s. `hpi`
|
|
636
|
+
// counted every one against `HPI_accuracy`, so the gate CI reads was partly
|
|
637
|
+
// measuring SpringBoard's stability.
|
|
638
|
+
//
|
|
639
|
+
// This is item 172 in the other column. There, `HPI_time` took every run's
|
|
640
|
+
// wall clock regardless of completion, so breaking a flow registered as the
|
|
641
|
+
// agent getting quicker; the fix was to time only completed runs. Accuracy
|
|
642
|
+
// had the mirror-image fault and kept it.
|
|
643
|
+
//
|
|
644
|
+
// Derived from the escalations this run already wrote — `device_cause` is
|
|
645
|
+
// set at the step that suffered it — so nothing new has to be plumbed and a
|
|
646
|
+
// run cannot claim a device fault that its own log does not show.
|
|
647
|
+
device_cause: escalations.map((e) => e.device_cause).find(Boolean) ?? null,
|
|
549
648
|
};
|
|
550
649
|
}
|
|
551
650
|
|
|
@@ -691,8 +790,14 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
691
790
|
};
|
|
692
791
|
}).sort((a, b) => a.flow.localeCompare(b.flow));
|
|
693
792
|
|
|
694
|
-
|
|
695
|
-
|
|
793
|
+
// A run the simulator broke is not a run the code got wrong. It is an
|
|
794
|
+
// **unmeasured** run, and it leaves the denominator rather than lowering the
|
|
795
|
+
// score — the same discipline as timing only completed runs (172), and the
|
|
796
|
+
// same discipline as the suite refusing to publish a partial HPI at all.
|
|
797
|
+
const lostToDevice = flows.filter((f) => f.device_cause && !f.completed);
|
|
798
|
+
const measurable = flows.filter((f) => !(f.device_cause && !f.completed));
|
|
799
|
+
const total = measurable.length;
|
|
800
|
+
const clean = measurable.filter((f) => f.completed && !f.wrong_action_taken).length;
|
|
696
801
|
const accuracy = total ? Number((clean / total).toFixed(3)) : null;
|
|
697
802
|
const times = perFlow.map((f) => f.hpi_time).filter((x) => Number.isFinite(x));
|
|
698
803
|
const hpiTime = harmonicMean(times);
|
|
@@ -700,6 +805,11 @@ export function hpi({ flows, baselines = {} }) {
|
|
|
700
805
|
flows: perFlow,
|
|
701
806
|
overall: {
|
|
702
807
|
runs: total,
|
|
808
|
+
// Said out loud, because a denominator that quietly shrinks is worse than
|
|
809
|
+
// one that is wrong: an accuracy of 1.0 over two measurable runs is not
|
|
810
|
+
// the same claim as 1.0 over eighteen.
|
|
811
|
+
runs_lost_to_device: lostToDevice.length,
|
|
812
|
+
device_causes: [...new Set(lostToDevice.map((f) => f.device_cause))],
|
|
703
813
|
flows_measured: perFlow.length,
|
|
704
814
|
flows_with_human_baseline: times.length,
|
|
705
815
|
hpi_accuracy: accuracy,
|
|
@@ -728,12 +838,20 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
728
838
|
const classifiedByReason = {};
|
|
729
839
|
const assumedByReason = {};
|
|
730
840
|
for (const r of REASONS) { byReason[r] = 0; classifiedByReason[r] = 0; assumedByReason[r] = 0; }
|
|
841
|
+
// Two more groupings, because the reason alone could not steer. `byVerdict`
|
|
842
|
+
// splits the largest class into the three different things it holds, and
|
|
843
|
+
// `byDevice` takes out the records that are the simulator rather than the
|
|
844
|
+
// code — those were being counted towards a perception phase.
|
|
845
|
+
const byVerdict = new Map();
|
|
846
|
+
const byDevice = new Map();
|
|
731
847
|
const byScreen = new Map();
|
|
732
848
|
const byOutcome = {};
|
|
733
849
|
const bySession = new Map();
|
|
734
850
|
const byFlow = new Map();
|
|
735
851
|
let avoidable = 0;
|
|
736
852
|
let unattributed = 0;
|
|
853
|
+
let derivedVerdicts = 0;
|
|
854
|
+
let derivedDevice = 0;
|
|
737
855
|
// Filtering happens here rather than at the call site so `total` and every
|
|
738
856
|
// rate below it describe the same set of records.
|
|
739
857
|
const kept = records.filter((r) => (session ? r?.session_id === session : true))
|
|
@@ -757,6 +875,18 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
757
875
|
// real one gets ignored, which this file already knows in another place.
|
|
758
876
|
if (r.classified === true) classifiedByReason[r.reason] += 1;
|
|
759
877
|
else if (r.classified === false) assumedByReason[r.reason] += 1;
|
|
878
|
+
// Derived for the records that predate the fields, rather than waiting for
|
|
879
|
+
// a fresh corpus. `detail` already carries both facts — the verdict as its
|
|
880
|
+
// prefix, the device signature inside the message — so a read-time
|
|
881
|
+
// derivation turns 1022 existing records into signal without rewriting a
|
|
882
|
+
// single line of the log. Marked in the output as derived, because a
|
|
883
|
+
// recorded fact and a parsed one are not the same evidence.
|
|
884
|
+
const verdict = r.verdict ?? verdictFromDetail(r.detail);
|
|
885
|
+
const device = r.device_cause ?? deviceCause(r.detail);
|
|
886
|
+
if (verdict) byVerdict.set(verdict, (byVerdict.get(verdict) ?? 0) + 1);
|
|
887
|
+
if (device) byDevice.set(device, (byDevice.get(device) ?? 0) + 1);
|
|
888
|
+
if (!r.verdict && verdict) derivedVerdicts += 1;
|
|
889
|
+
if (!r.device_cause && device) derivedDevice += 1;
|
|
760
890
|
byOutcome[r.outcome] = (byOutcome[r.outcome] ?? 0) + 1;
|
|
761
891
|
// Already avoided locally, so not avoidable by anything unbuilt.
|
|
762
892
|
if (r.outcome !== 'resolved_locally') avoidable += 1;
|
|
@@ -770,8 +900,19 @@ export function breakdown(records, { session = null, flow = null } = {}) {
|
|
|
770
900
|
return { session_id: id, client, count };
|
|
771
901
|
})
|
|
772
902
|
.sort((a, b) => b.count - a.count);
|
|
903
|
+
const sorted = (m) => [...m.entries()].sort((a, b) => b[1] - a[1]).map(([name, count]) => ({ name, count }));
|
|
773
904
|
return {
|
|
774
905
|
total,
|
|
906
|
+
// What the reason could not say. `verdicts` is the largest class split into
|
|
907
|
+
// the faculties it actually implies; `device` is the part that is item 173
|
|
908
|
+
// wearing a code reason.
|
|
909
|
+
verdicts: sorted(byVerdict),
|
|
910
|
+
device: sorted(byDevice),
|
|
911
|
+
device_total: [...byDevice.values()].reduce((a, b) => a + b, 0),
|
|
912
|
+
// How much of the two groupings above was parsed out of `detail` rather
|
|
913
|
+
// than recorded at the time. A reader deciding a phase order should know
|
|
914
|
+
// which half they are looking at.
|
|
915
|
+
derived: { verdicts: derivedVerdicts, device: derivedDevice },
|
|
775
916
|
// The log is per-device and shared: two agents on one booted simulator
|
|
776
917
|
// write one interleaved file. More than one session here means the counts
|
|
777
918
|
// below are a pool, and CLAUDE.md uses those counts to choose a phase.
|
package/src/navigate.js
CHANGED
|
@@ -47,6 +47,16 @@ function refuse(udid, result, { detail = null, flowName = null } = {}) {
|
|
|
47
47
|
if (reason) {
|
|
48
48
|
metrics.recordEscalation(udid, {
|
|
49
49
|
reason,
|
|
50
|
+
// Read, not assumed. `recordEscalation` defaults `classified` to false
|
|
51
|
+
// so a careless caller cannot claim precision it does not have, which
|
|
52
|
+
// is right — and this caller is not careless: `result.reason` is a
|
|
53
|
+
// named refusal (`no-route`, `unreplayable-edge`, `unknown-flow`,
|
|
54
|
+
// `arrived-elsewhere`…) and PLAN_REASONS maps it deterministically.
|
|
55
|
+
// Saying nothing filed 82 records on the bench device as "reason
|
|
56
|
+
// assumed" when the reason was known exactly, which makes the log
|
|
57
|
+
// understate its own knowledge and the report decline to name a
|
|
58
|
+
// faculty it was entitled to name.
|
|
59
|
+
classified: true,
|
|
50
60
|
// A refusal by `goto` is about a destination and one by `flow run` is
|
|
51
61
|
// about a named flow. Either is what a breakdown wants to group by.
|
|
52
62
|
flowName,
|
package/src/wedge.js
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
// What is this device actually doing right now?
|
|
2
|
+
//
|
|
3
|
+
// `doctor` answers "can this machine capture". This answers a different
|
|
4
|
+
// question, and it is the one item 173 has never been able to answer: when a
|
|
5
|
+
// device stops presenting the app, *which* of several failures is it?
|
|
6
|
+
//
|
|
7
|
+
// The reason this exists is that the evidence has been arriving as a
|
|
8
|
+
// consequence rather than an observation. `scripts/device-state.mjs` recognises
|
|
9
|
+
// a wedge from the shape of a tour's failure text — "two labels, one of them a
|
|
10
|
+
// clock, on a still screen" — and names it "a launched app never came to the
|
|
11
|
+
// front". That classification is useful for deciding whether to revive, and it
|
|
12
|
+
// is not a diagnosis: it cannot tell a dead framebuffer from a device that is
|
|
13
|
+
// genuinely showing a lock screen from an app that never fronted. Three
|
|
14
|
+
// hypotheses, one symptom, and a regex standing where a measurement should be.
|
|
15
|
+
//
|
|
16
|
+
// So the facts get gathered at the moment of the wedge instead of inferred
|
|
17
|
+
// afterwards, and the classifier over them is a **pure function** with fixtures.
|
|
18
|
+
// That part is deliberate: `device-state.mjs` records that "two runtime bugs in
|
|
19
|
+
// this project came from logic that was correct and had never executed",
|
|
20
|
+
// because the only thing exercising it was a hosted runner at minute fifteen of
|
|
21
|
+
// a job. Nothing here needs a device to be tested.
|
|
22
|
+
import * as api from './index.js';
|
|
23
|
+
import * as frontmost from './frontmost.js';
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* How little agreement between the sensors counts as "these are different
|
|
27
|
+
* screens".
|
|
28
|
+
*
|
|
29
|
+
* **Measured, and the first version of this constant was justified with the
|
|
30
|
+
* wrong number.** It cited CLAUDE.md's "the tree and OCR agree on 0.33–0.47",
|
|
31
|
+
* which is a figure about *structural tokens* — a different quantity from the
|
|
32
|
+
* element-level fusion counted here. Printing a healthy band of 0.33–0.47 next
|
|
33
|
+
* to a live reading of 0.846 is how that got noticed.
|
|
34
|
+
*
|
|
35
|
+
* Sampled on `326464A4` (iPhone 17 Pro, iOS 26.5) across five real screens —
|
|
36
|
+
* Settings root, General, About, springboard, Reminders:
|
|
37
|
+
*
|
|
38
|
+
* 0.857 0.833 0.929 0.846 0.667 median 0.846
|
|
39
|
+
*
|
|
40
|
+
* So healthy element fusion is **0.67-0.93** (stated as 0.66 so a reading at the floor does not print as below it), and the floor belongs to the
|
|
41
|
+
* sparsest screen, which is also the case `ENOUGH_TO_COMPARE` withholds
|
|
42
|
+
* judgement on. A threshold of 0.1 sits 6.7x below the observed floor: this is
|
|
43
|
+
* not a band inside the metric's own noise, which is the mistake the HPI_time
|
|
44
|
+
* gate made.
|
|
45
|
+
*
|
|
46
|
+
* What the collapse means: the tree is read live and in-process, OCR reads a
|
|
47
|
+
* framebuffer that can go stale without saying so. If both sensors report
|
|
48
|
+
* plenty and almost nothing fuses, they are looking at different screens, and
|
|
49
|
+
* the frame is the one that is behind. A peer hit exactly this — `sim_look`
|
|
50
|
+
* returned a web form from an earlier session on the device while the element
|
|
51
|
+
* map, taken at the same moment, correctly described the app in front of them.
|
|
52
|
+
*/
|
|
53
|
+
export const DISAGREEMENT = 0.1;
|
|
54
|
+
|
|
55
|
+
/** The range observed on healthy screens, for the report to print honestly. */
|
|
56
|
+
export const HEALTHY_FUSION = '0.66-0.93';
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Both sensors need at least this many elements before their disagreement means
|
|
60
|
+
* anything. A screen with one label from each cannot be said to disagree, and a
|
|
61
|
+
* springboard or a lock screen is legitimately sparse.
|
|
62
|
+
*/
|
|
63
|
+
export const ENOUGH_TO_COMPARE = 3;
|
|
64
|
+
|
|
65
|
+
/** At or below this from both sensors, there is effectively nothing on screen. */
|
|
66
|
+
export const SPARSE = 2;
|
|
67
|
+
|
|
68
|
+
const seenBy = (targets, sensor) =>
|
|
69
|
+
targets.filter((t) => String(t.source ?? '').split('|').includes(sensor)).length;
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Read everything that discriminates between the failure modes, in one pass.
|
|
73
|
+
*
|
|
74
|
+
* Every field here exists because it separates two hypotheses. Nothing is
|
|
75
|
+
* collected because it is interesting.
|
|
76
|
+
*/
|
|
77
|
+
export async function snapshot(deviceQuery, { options } = {}) {
|
|
78
|
+
// **`ensureDaemon` throws on the loudest condition this module exists to
|
|
79
|
+
// name.** "the daemon is running and the display produced no frame in 60s" is
|
|
80
|
+
// reported as an exception, so the first version of this function propagated
|
|
81
|
+
// it and `simframe diagnose` died with a stack trace on a genuinely wedged
|
|
82
|
+
// device — the one moment it is worth running. Caught within an hour of
|
|
83
|
+
// shipping, by the device wedging.
|
|
84
|
+
//
|
|
85
|
+
// So a snapshot of a dead device is a snapshot, not an error. `classify`
|
|
86
|
+
// already has a verdict for it.
|
|
87
|
+
let device;
|
|
88
|
+
let state = null;
|
|
89
|
+
let daemonError = null;
|
|
90
|
+
try {
|
|
91
|
+
({ device, state } = await api.ensureDaemon(deviceQuery, options));
|
|
92
|
+
} catch (err) {
|
|
93
|
+
daemonError = err.message;
|
|
94
|
+
device = { udid: String(deviceQuery ?? '?'), name: String(deviceQuery ?? 'unknown device') };
|
|
95
|
+
}
|
|
96
|
+
const udid = device.udid;
|
|
97
|
+
if (daemonError) {
|
|
98
|
+
return {
|
|
99
|
+
device: { udid, name: device.name },
|
|
100
|
+
readError: daemonError,
|
|
101
|
+
frame: null,
|
|
102
|
+
elements: { total: 0, ax: 0, ocr: 0, fused: 0 },
|
|
103
|
+
agreement: null,
|
|
104
|
+
frontmost: null,
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
let identity = null;
|
|
109
|
+
let readError = null;
|
|
110
|
+
try {
|
|
111
|
+
identity = await api.screenIdentity(udid, { options, confirmNovel: false });
|
|
112
|
+
} catch (err) {
|
|
113
|
+
readError = err.message;
|
|
114
|
+
}
|
|
115
|
+
const targets = identity?.entry?.targets ?? [];
|
|
116
|
+
|
|
117
|
+
// Who holds the front, by pid, which is item 169's contribution and the only
|
|
118
|
+
// sensor here that does not go through the display at all.
|
|
119
|
+
let front = null;
|
|
120
|
+
try {
|
|
121
|
+
front = await frontmost.read(udid);
|
|
122
|
+
} catch (err) {
|
|
123
|
+
front = { pid: null, title: null, error: err.message };
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const ax = seenBy(targets, 'ax');
|
|
127
|
+
const ocr = seenBy(targets, 'ocr');
|
|
128
|
+
const fused = targets.filter((t) => {
|
|
129
|
+
const parts = String(t.source ?? '').split('|');
|
|
130
|
+
return parts.includes('ax') && parts.includes('ocr');
|
|
131
|
+
}).length;
|
|
132
|
+
|
|
133
|
+
return {
|
|
134
|
+
device: { udid, name: device.name },
|
|
135
|
+
readError,
|
|
136
|
+
frame: state
|
|
137
|
+
? {
|
|
138
|
+
seq: state.seq ?? null,
|
|
139
|
+
ageMs: state.capturedAt ? Date.now() - state.capturedAt : null,
|
|
140
|
+
stableForMs: state.stableForMs ?? null,
|
|
141
|
+
size: state.width && state.height ? `${state.width}x${state.height}` : null,
|
|
142
|
+
hash: state.hash ? String(state.hash).slice(0, 8) : null,
|
|
143
|
+
}
|
|
144
|
+
: null,
|
|
145
|
+
elements: { total: targets.length, ax, ocr, fused },
|
|
146
|
+
// Null rather than a number when it cannot be computed, so a caller cannot
|
|
147
|
+
// read "0 agreement" off a screen nobody could compare.
|
|
148
|
+
agreement: ax > 0 && ocr > 0 ? Number((fused / Math.min(ax, ocr)).toFixed(3)) : null,
|
|
149
|
+
frontmost: front,
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Name the state, or say it looks fine.
|
|
155
|
+
*
|
|
156
|
+
* Ordered most specific first, and each verdict says what it is evidence *of*
|
|
157
|
+
* rather than what to do about it — the same discipline as the escalation
|
|
158
|
+
* reasons, where a vocabulary that admits "other" collects a pile of "other".
|
|
159
|
+
*/
|
|
160
|
+
export function classify(snap) {
|
|
161
|
+
if (!snap?.frame) {
|
|
162
|
+
return {
|
|
163
|
+
state: 'capture-down',
|
|
164
|
+
// The daemon's own words when it has them. They are more specific than
|
|
165
|
+
// anything derivable here — "produced no frame in 60s" distinguishes a
|
|
166
|
+
// live daemon over a dead display from a daemon that is not running.
|
|
167
|
+
detail: snap?.readError
|
|
168
|
+
? `capture is not producing frames: ${snap.readError}`
|
|
169
|
+
: 'no frame state at all — the daemon is not producing frames',
|
|
170
|
+
revive: true,
|
|
171
|
+
};
|
|
172
|
+
}
|
|
173
|
+
const { total, ax, ocr } = snap.elements ?? { total: 0, ax: 0, ocr: 0 };
|
|
174
|
+
if (snap.readError) {
|
|
175
|
+
return { state: 'read-failed', detail: `the screen could not be read: ${snap.readError}`, revive: true };
|
|
176
|
+
}
|
|
177
|
+
if (total === 0) {
|
|
178
|
+
return {
|
|
179
|
+
state: 'nothing-readable',
|
|
180
|
+
detail: 'neither the accessibility tree nor OCR found a single element',
|
|
181
|
+
revive: true,
|
|
182
|
+
};
|
|
183
|
+
}
|
|
184
|
+
// The two sensors are describing different screens. The tree is read live and
|
|
185
|
+
// in-process; OCR reads the framebuffer. So the frame is the stale one.
|
|
186
|
+
if (ax >= ENOUGH_TO_COMPARE && ocr >= ENOUGH_TO_COMPARE
|
|
187
|
+
&& snap.agreement != null && snap.agreement <= DISAGREEMENT) {
|
|
188
|
+
return {
|
|
189
|
+
state: 'stale-frame',
|
|
190
|
+
detail: `the tree found ${ax} element(s) and OCR found ${ocr}, and they fuse on `
|
|
191
|
+
+ `${snap.agreement} of them (measured healthy: ${HEALTHY_FUSION}). The tree is read live, `
|
|
192
|
+
+ 'so the framebuffer is the one that is behind — an image from this device is not safe to trust',
|
|
193
|
+
revive: true,
|
|
194
|
+
};
|
|
195
|
+
}
|
|
196
|
+
// An app holds the front, and the display is showing almost nothing. This is
|
|
197
|
+
// the CI signature that `device-state.mjs` recognises as "a clock and nothing
|
|
198
|
+
// else", now stated as an observation. It deliberately does NOT claim to know
|
|
199
|
+
// whether this is a lock screen, a dead surface or a crashed SpringBoard —
|
|
200
|
+
// that is the next question, and pretending to answer it is what put a regex
|
|
201
|
+
// where a measurement belongs.
|
|
202
|
+
if (snap.frontmost?.pid && ax <= SPARSE && ocr <= SPARSE) {
|
|
203
|
+
return {
|
|
204
|
+
state: 'not-presenting',
|
|
205
|
+
detail: `pid ${snap.frontmost.pid}`
|
|
206
|
+
+ `${snap.frontmost.title ? ` (${snap.frontmost.title})` : ''} holds the front, but the `
|
|
207
|
+
+ `display shows ${total} element(s). Something is in front of the app, or the display is `
|
|
208
|
+
+ 'not painting it. Which of those it is, is not knowable from here',
|
|
209
|
+
revive: true,
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
return { state: 'healthy', detail: `${total} element(s), sensors agree on ${snap.agreement ?? 'n/a'}`, revive: false };
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
export async function diagnose(deviceQuery, { options } = {}) {
|
|
216
|
+
const snap = await snapshot(deviceQuery, { options });
|
|
217
|
+
return { ...snap, verdict: classify(snap) };
|
|
218
|
+
}
|