simframe 0.12.0 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "simframe",
3
- "version": "0.12.0",
3
+ "version": "0.12.1",
4
4
  "mcpName": "io.github.lvlrSajjad/simframe",
5
5
  "description": "Always-warm iOS Simulator and Android emulator frames: agents read the screen in ~20ms instead of waiting on screenshots. MCP server + CLI.",
6
6
  "keywords": [
@@ -0,0 +1,107 @@
1
+ #!/usr/bin/env node
2
+ // Score a ruling population against the baselines that could embarrass it.
3
+ //
4
+ // Written **before** the population it first scored was finished, deliberately.
5
+ // The previous attempt computed a `stillMs` threshold after seeing the answers,
6
+ // on 14 samples of which 12 shared one label, and produced "100%" — a number
7
+ // that was fitted, not measured. Fixing that afterwards is not possible: you
8
+ // cannot un-see the data. So the split rule, the candidate thresholds and the
9
+ // baselines are all fixed here in advance.
10
+ //
11
+ // node scripts/score-rulings.mjs --device=<udid>
12
+ //
13
+ // What it will not do: pick the best threshold and report its score. It fits on
14
+ // the first half and scores on the second, and prints both numbers so a gap
15
+ // between them is visible as overfitting rather than hidden as success.
16
+ import * as metrics from '../src/metrics.js';
17
+ import { resolveDevice } from '../src/platform/index.js';
18
+
19
+ const arg = (n, d) => {
20
+ const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
21
+ return hit ? hit.slice(n.length + 3) : d;
22
+ };
23
+
24
+ /** The fixtures, and the word each one's situation actually calls for. */
25
+ const CORRECT = {
26
+ 'the list is still loading; its rows arrive shortly after launch': 'wait',
27
+ 'the detail screen is still fetching; its text arrives shortly': 'wait',
28
+ 'the list arrives in waves and this row is in the last one': 'wait',
29
+ 'Review is blocked until Species is filled in, and it is empty': 'stop',
30
+ 'the first submit always fails and the second works, so waiting cannot help': 'stop',
31
+ };
32
+
33
+ /** `retry` and `wait` differ only in how long they settle, so both satisfy a wait. */
34
+ const satisfies = (decision, want) =>
35
+ (want === 'wait' ? decision === 'wait' || decision === 'retry' : decision === want);
36
+
37
+ const dev = await resolveDevice(arg('device'));
38
+ const all = metrics.readSupervisions(dev.udid).filter((r) => CORRECT[r.expect]);
39
+ // `--last=N` scores one batch rather than the whole log. Needed the first time
40
+ // this ran: the log still held rulings from a deliberately skewed population,
41
+ // so the balance read 19/35 and the baseline 65% when the batch actually under
42
+ // test was 16/16. Mixing a known-bad population into the denominator is the
43
+ // same error as before wearing a different hat.
44
+ const lastN = Number(arg('last', 0));
45
+ const rows = lastN > 0 ? all.slice(-lastN) : all;
46
+ if (rows.length < 8) {
47
+ console.error(`only ${rows.length} labelled ruling(s) — run scripts/collect-rulings.mjs first`);
48
+ process.exit(2);
49
+ }
50
+
51
+ const want = (r) => CORRECT[r.expect];
52
+ const counts = rows.reduce((a, r) => ({ ...a, [want(r)]: (a[want(r)] ?? 0) + 1 }), {});
53
+ const commonest = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
54
+ const baseline = commonest[1] / rows.length;
55
+
56
+ console.log(`${rows.length} labelled rulings on ${dev.name}`);
57
+ console.log(`balance: ${JSON.stringify(counts)}`);
58
+ console.log(`majority-class baseline: always "${commonest[0]}" scores ${(100 * baseline).toFixed(0)}%`);
59
+ if (baseline > 0.65) {
60
+ console.log('\nSKEWED — any accuracy below is mostly a fact about the fixture set.');
61
+ }
62
+
63
+ const acc = (rs, predict) => rs.filter((r) => satisfies(predict(r), want(r))).length / (rs.length || 1);
64
+ const pct = (x) => `${(100 * x).toFixed(0)}%`;
65
+
66
+ console.log(`\n${'the model'.padEnd(30)} ${pct(acc(rows, (r) => r.decision))}`);
67
+ console.log(`${`always "${commonest[0]}"`.padEnd(30)} ${pct(baseline)}`);
68
+
69
+ // --- the stillMs rule, fitted on one half and scored on the other.
70
+ //
71
+ // Split by *recording order* rather than at random or by fixture, because the
72
+ // alternative is choosing a split, and choosing is the thing that went wrong.
73
+ const half = Math.floor(rows.length / 2);
74
+ const fit = rows.slice(0, half);
75
+ const held = rows.slice(half);
76
+ const CANDIDATES = [1000, 1500, 2000, 2500, 3000, 4000, 5000, 6000, 7000];
77
+ const rule = (t) => (r) => ((r.still_ms ?? 0) > t ? 'stop' : 'wait');
78
+
79
+ let best = null;
80
+ for (const t of CANDIDATES) {
81
+ const a = acc(fit, rule(t));
82
+ if (!best || a > best.a) best = { t, a };
83
+ }
84
+ console.log(`\nstillMs threshold, fitted on the first ${fit.length} and scored on the last ${held.length}:`);
85
+ console.log(` chosen on the fit half: > ${best.t}ms (${pct(best.a)} there)`);
86
+ console.log(` ${'on the held-out half:'.padEnd(24)} ${pct(acc(held, rule(best.t)))}`);
87
+ console.log(` ${'the model, same half:'.padEnd(24)} ${pct(acc(held, (r) => r.decision))}`);
88
+ console.log('\nA rule that scores far better on the fit half than the held-out half was');
89
+ console.log('fitted to noise. That gap is the number this script exists to print.');
90
+
91
+ // --- the direction of the errors, which does not depend on the balance at all.
92
+ const wrong = rows.filter((r) => !satisfies(r.decision, want(r)));
93
+ const dirs = wrong.reduce((a, r) => {
94
+ const k = `said "${r.decision}" where "${want(r)}" was right`;
95
+ return { ...a, [k]: (a[k] ?? 0) + 1 };
96
+ }, {});
97
+ console.log(`\n${wrong.length} error(s) of ${rows.length}:`);
98
+ for (const [k, n] of Object.entries(dirs).sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${k}`);
99
+ if (Object.keys(dirs).length === 1 && wrong.length > 2) {
100
+ console.log('\nAll errors in one direction. A one-directional bias is what an abstain');
101
+ console.log('token addresses (DEFERRED 100), whatever the headline accuracy says.');
102
+ }
103
+
104
+ const lat = rows.map((r) => r.latency_ms).filter(Number.isFinite).sort((a, b) => a - b);
105
+ if (lat.length) console.log(`\nmedian judgement latency: ${lat[lat.length >> 1]}ms`);
106
+ const timed = rows.filter((r) => Number.isFinite(r.edge_p95_ms)).length;
107
+ console.log(`edges the graph had timed: ${timed}/${rows.length}`);
@@ -66,6 +66,19 @@ async function bootedDevices(opts) {
66
66
  */
67
67
  async function resolveDevice(query, opts) {
68
68
  const all = await listDevices(opts);
69
+ return pickDevice(query, all);
70
+ }
71
+
72
+ /**
73
+ * Which device a query means, given the whole list.
74
+ *
75
+ * Separated from the listing so the decision can be **tested** rather than
76
+ * reasoned about, because two peer rounds in a row reported being handed the
77
+ * wrong device and both were decided here. The same move as `decisionOf` in the
78
+ * supervisor: the one line where a wrong answer is expensive should be a
79
+ * function somebody can call with adversarial input.
80
+ */
81
+ export function pickDevice(query, all) {
69
82
  const booted = all.filter((d) => d.state === 'Booted');
70
83
  if (!query) {
71
84
  if (booted.length === 0) throw new Error('no booted simulator (open Simulator.app or run `xcrun simctl boot <udid>`)');
@@ -98,8 +111,40 @@ async function resolveDevice(query, opts) {
98
111
  const q = query.toLowerCase();
99
112
  const pools = [booted, all];
100
113
  for (const pool of pools) {
101
- const exact = pool.find((d) => d.udid.toLowerCase() === q || d.name.toLowerCase() === q);
102
- if (exact) return exact;
114
+ // A UDID is unique, so an exact UDID match needs no further thought.
115
+ const byUdid = pool.find((d) => d.udid.toLowerCase() === q);
116
+ if (byUdid) return byUdid;
117
+ // A **name is not unique**, and this branch used to treat it as though it
118
+ // were: one `find` over both fields returned whichever device the list
119
+ // happened to put first, and short-circuited past the ambiguity guard
120
+ // below. Item 83 recorded that two booted devices on this machine are both
121
+ // called "iPhone 17 Pro" and answered it by warning in `sim_devices` and
122
+ // printing a UDID prefix in headers — leaving the resolver, which is where
123
+ // the choice is actually made, untouched.
124
+ //
125
+ // What that cost, reported from a three-hour session on a real app: a
126
+ // caller passed the shared name, read a screen that was "38ms old" and an
127
+ // hour wrong, concluded the app had signed itself out, and abandoned a
128
+ // verification run that was fine. The frame was fresh — it was the *other*
129
+ // device's, idling on a login screen. `refresh: true` returned the matching
130
+ // tree because it refreshed the same wrong device. Two independent-looking
131
+ // sources agreeing with each other and both wrong.
132
+ //
133
+ // So a name that names two devices is an ambiguity, exactly like a partial
134
+ // match that hits two, and it refuses for the same reason: the cost of
135
+ // guessing wrong is reading somebody else's screen and believing it.
136
+ const byName = pool.filter((d) => d.name.toLowerCase() === q);
137
+ if (byName.length === 1) return byName[0];
138
+ if (byName.length > 1) {
139
+ throw Object.assign(
140
+ new Error(
141
+ `"${query}" is the name of ${byName.length} devices: `
142
+ + `${byName.map((d) => d.udid).join(', ')} — a name cannot say which one you mean, `
143
+ + 'so pass the UDID',
144
+ ),
145
+ { ambiguous: true },
146
+ );
147
+ }
103
148
  const partial = pool.filter((d) => d.name.toLowerCase().includes(q));
104
149
  if (partial.length === 1) return partial[0];
105
150
  if (partial.length > 1) {