simframe 0.12.1 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -50,6 +50,9 @@ let failures = 0;
50
50
  */
51
51
  let deviceDied = null;
52
52
 
53
+ /** EX_TEMPFAIL: the device died under the checks, so nothing was tested. */
54
+ const DEVICE_DIED_EXIT = 75;
55
+
53
56
  function check(ok, label, detail = '') {
54
57
  if (!ok) failures += 1;
55
58
  console.log(`${ok ? 'ok ' : 'FAIL'} ${label}${detail ? ` — ${detail}` : ''}`);
@@ -171,9 +174,21 @@ async function jsonRetry(args, opts, attempts = 3) {
171
174
  console.error(` ${String(last.message).split('\n')[0]}`);
172
175
  console.error('\nEverything after this point would be testing a dead simulator, so the run');
173
176
  console.error('stops here. This is not a memory-layer failure — it is the device-state');
174
- console.error('problem in docs/DEFERRED.md. A device restart is the only known cure;');
175
- console.error('on a hosted runner it means a retry.');
176
- process.exit(1);
177
+ console.error('problem in docs/DEFERRED.md (126). A device restart is the only known cure,');
178
+ console.error('and `simframe revive` is that restart.');
179
+ // Exit 75, not 1, and the distinction is the whole point of this file.
180
+ //
181
+ // "A check about the memory layer failed" and "the simulator died under the
182
+ // checks" are different conditions with different responses, and for two CI
183
+ // rounds they were one exit code — so a caller could only retry everything
184
+ // or retry nothing. 75 is EX_TEMPFAIL, which is exactly what this is: the
185
+ // subject under test was never reached.
186
+ //
187
+ // The caller reviving and running again is not papering over a product bug.
188
+ // The wedge is a documented CoreSimulator condition, `frame --fresh` names
189
+ // it, and `revive` is the cure this project ships for it — CI simply had no
190
+ // way to say "use it".
191
+ process.exit(DEVICE_DIED_EXIT);
177
192
  }
178
193
  throw last;
179
194
  }
@@ -244,7 +259,36 @@ if (failures) {
244
259
 
245
260
  console.log('\n--- the screen map ---');
246
261
  await jsonRetry(['do', LOOP], { allowFail: true });
247
- const map = await jsonRetry(['ui']);
262
+ const map = await readableMap();
263
+
264
+ /**
265
+ * A map that is empty *and* says a sensor failed is a blink, not a result.
266
+ *
267
+ * `jsonRetry` retries a command that throws, and this one did not throw: `ui`
268
+ * returned 200 with zero elements and a degraded note, which is the shape of
269
+ * failure this whole repo keeps writing rules about. The run that made this
270
+ * necessary had the accessibility read time out once, report an empty map, and
271
+ * then resolve a ref correctly sixty seconds later on the same device — so the
272
+ * device was fine and the check had caught one bad read.
273
+ *
274
+ * Deliberately narrow. An empty map with **no** degraded sensor is a real
275
+ * answer — that is a blank screen and the check should fail on it. Only an
276
+ * empty map that admits a layer did not answer is worth asking again, and after
277
+ * three attempts it fails with what the sensor said, which is the diagnosis
278
+ * either way.
279
+ */
280
+ async function readableMap(attempts = 3) {
281
+ let last;
282
+ for (let i = 0; i < attempts; i += 1) {
283
+ last = await jsonRetry(['ui']);
284
+ if (last.elements?.length || !(last.degraded ?? []).length) return last;
285
+ if (i < attempts - 1) {
286
+ console.log(` (the map came back empty and a sensor said why — retrying \`ui\`: ${(last.degraded ?? []).join('; ')})`);
287
+ await new Promise((r) => setTimeout(r, 2000));
288
+ }
289
+ }
290
+ return last;
291
+ }
248
292
 
249
293
  // Report what actually answered rather than asserting the runner's situation.
250
294
  // This line used to read "with no accessibility tree available" unconditionally,
@@ -263,14 +307,38 @@ check(Number.isFinite(map.points?.width) && Number.isFinite(map.points?.height),
263
307
  'the map knows the screen size in points', `${map.points?.width}x${map.points?.height}pt`);
264
308
 
265
309
  const refs = (map.elements ?? []).map((e) => e.ref);
266
- check(refs.every((r, i) => r === i + 1),
267
- 'refs are numbered 1..n with no gaps', `#1..#${refs.length}`);
268
- check((map.elements ?? []).every((e) =>
269
- Number.isInteger(e.x) && Number.isInteger(e.y)
270
- && e.y >= 0 && e.y <= map.points.height && e.x >= 0 && e.x <= map.points.width),
271
- 'every element has a tap point on the screen');
272
- check(!(map.elements ?? []).some((e) => e.region === 'status-bar'),
273
- 'the status bar is not offered as something to tap');
310
+ // Three checks over a collection, and `every`/`!some` are all true of an empty
311
+ // one. On a CI run whose map came back with **0 elements** they printed
312
+ // `ok refs are numbered 1..n with no gaps — #1..#0` and two more like it:
313
+ // three lines of reassurance about nothing, directly under the failure that
314
+ // said the map was empty.
315
+ //
316
+ // That is the exact defect three field reports spent a day describing — a
317
+ // confident statement that verified nothing — and the harness was doing it to
318
+ // itself in the same output. Untested is not passed.
319
+ if (!map.elements?.length) {
320
+ for (const label of [
321
+ 'refs are numbered 1..n with no gaps',
322
+ 'every element has a tap point on the screen',
323
+ 'the status bar is not offered as something to tap',
324
+ ]) skip(label, 'the map was empty, so there was nothing to check');
325
+ // And say what the device was showing, because an empty map on a live device
326
+ // is the shape this project has chased under four different symptoms. The
327
+ // liveness note carries the ignored-gesture check; `stableForMs` and the
328
+ // frame age are the two numbers whose disagreement names a dead surface.
329
+ const st = await jsonRetry(['state'], { allowFail: true });
330
+ console.log(` the device at that moment: frame #${st?.seq ?? '?'}, `
331
+ + `${st?.stableForMs ?? '?'}ms still, ${st?.live?.note ?? 'liveness reported nothing'}`);
332
+ } else {
333
+ check(refs.every((r, i) => r === i + 1),
334
+ 'refs are numbered 1..n with no gaps', `#1..#${refs.length}`);
335
+ check(map.elements.every((e) =>
336
+ Number.isInteger(e.x) && Number.isInteger(e.y)
337
+ && e.y >= 0 && e.y <= map.points.height && e.x >= 0 && e.x <= map.points.width),
338
+ 'every element has a tap point on the screen');
339
+ check(!map.elements.some((e) => e.region === 'status-bar'),
340
+ 'the status bar is not offered as something to tap');
341
+ }
274
342
 
275
343
  console.log('\n--- element refs ---');
276
344
  // Re-read on a screen we chose, rather than on whatever the device happened to
@@ -296,7 +364,8 @@ if (first) {
296
364
  // hashes only have to agree with that, and they only get a say when they are
297
365
  // informative enough to have one.
298
366
  const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
299
- if (!ran(left)) skip('the screen actually changed before testing the stale ref',
367
+ const launched = ran(left);
368
+ if (!launched) skip('the screen actually changed before testing the stale ref',
300
369
  'the second app never launched, so there was no screen change to test against');
301
370
  else check(moved, 'the screen actually changed before testing the stale ref',
302
371
  `${before.slice(0, 10)} -> ${after.slice(0, 10)}`
@@ -305,7 +374,16 @@ if (first) {
305
374
  // reports "the stale-ref guard failed" for a device that never left the
306
375
  // screen, which is a false accusation against the one layer this file exists
307
376
  // to defend — and it is how this check has failed twice.
308
- if (moved) {
377
+ // A skip has to propagate. The precondition above reported NOT TESTED and this
378
+ // check ran anyway and failed — which is the harness doing to itself, one line
379
+ // later, exactly what `skip` was written to stop it doing. `moved` is true
380
+ // when a hash is too degenerate to have a say, and that is right for "did the
381
+ // screen change" and wrong as a licence to run a check whose setup is known
382
+ // not to have happened.
383
+ if (!launched) {
384
+ skip('a ref numbered on another screen refuses instead of tapping those coordinates',
385
+ 'we never reached another screen, so there was nothing to refuse from');
386
+ } else if (moved) {
309
387
  // This matched on prose twice and went red twice, both times for a refusal
310
388
  // that was correct and better worded than the alternation knew — most
311
389
  // recently `"Welcome to Reminders" is not on this screen`, which refuses
@@ -491,7 +569,12 @@ const walked = await jsonRetry(['goto', target.hash], { allowFail: true });
491
569
  // neither walking there nor naming why it cannot, so this check failed with an
492
570
  // empty detail — the reason was `undefined` — and the check was right to fail.
493
571
  // Now there is a name for it, and the list has to know the name.
494
- const outcomes = ['no-route', 'unreplayable-edge', 'ambiguous', 'unknown-screen', 'no-identity'];
572
+ // `route-halted` and `arrived-elsewhere` are new: a walk that ran and did not
573
+ // land used to return `{ok: false}` with no reason at all, which failed this
574
+ // very check with an empty detail. It was the one outcome here nobody had
575
+ // named, and the check found it.
576
+ const outcomes = ['no-route', 'unreplayable-edge', 'ambiguous', 'unknown-screen', 'no-identity',
577
+ 'route-halted', 'arrived-elsewhere'];
495
578
  check(walked.ok === true || outcomes.includes(walked.reason),
496
579
  'and asked for a screen it knows, it either walks there or names why it cannot',
497
580
  walked.ok ? (walked.already ? 'already there' : `walked ${walked.ranSteps} step(s)`) : walked.reason);
@@ -310,3 +310,17 @@ if (unattributed) console.log(`\n${unattributed} ruling(s) came from a step this
310
310
  }
311
311
  }
312
312
  console.log(`\nthe log now holds ${all.length} ruling(s) — simframe supervisions --device=${dev.udid}`);
313
+
314
+ // Close the helper, or this script never exits.
315
+ //
316
+ // The supervisor keeps a warm child process and deliberately does not unref its
317
+ // stdout — unreferencing it once unreferenced the pipe every request waits on,
318
+ // and the process then exited silently mid-await. The consequence nobody had
319
+ // noticed is at *this* end: the script finishes its work, prints this line, and
320
+ // then sits at 0% CPU forever with the helper idle at `readLine`, because the
321
+ // live child is holding the event loop open.
322
+ //
323
+ // That is the whole story of three "orphaned" collect-rulings processes killed
324
+ // by PID the night before, which were attributed to a task runner not killing
325
+ // its children. They had each finished their work and could not leave.
326
+ supervisor.close();
@@ -129,7 +129,24 @@ for (let round = 1; round <= rounds; round += 1) {
129
129
  if (screen.steps?.length) {
130
130
  // Verification is off: this measures fingerprints, and a wrong-turn
131
131
  // verdict computed from the very tokens under test would be circular.
132
- await actions.runScript(device, { steps: screen.steps, verify: false });
132
+ //
133
+ // The tour asserts arrival rather than sleeping through it, so a step
134
+ // here can now fail — `openUrl` returns NSPOSIXErrorDomain 60 on a loaded
135
+ // hosted runner, and a page that never renders no longer passes silently
136
+ // as a six-token reading. That is the trade this harness wants: a loud
137
+ // failure naming the screen, over a quiet one that shows up thirty lines
138
+ // later as a threshold with no clearance. Everything read so far is
139
+ // written out first, because a failing run is the one whose evidence
140
+ // matters.
141
+ try {
142
+ await actions.runScript(device, { steps: screen.steps, verify: false });
143
+ } catch (err) {
144
+ save({ abandonedAt: { screen: screen.name, round, error: err.message } });
145
+ console.error(`\nFAIL round ${round}, "${screen.name}" never arrived: ${err.message}`);
146
+ console.error('The tour waits for something each screen actually shows. This is that wait');
147
+ console.error('giving up — not a fingerprint result. Check the app, the network, or the runner.');
148
+ process.exit(1);
149
+ }
133
150
  }
134
151
  const id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
135
152
  // Did we actually arrive? Two differently-named screens reading the same
@@ -147,8 +164,21 @@ for (let round = 1; round <= rounds; round += 1) {
147
164
  // means the navigation did not happen.
148
165
  const s = fingerprint.similarity(previous.tokens, id.tokens ?? []);
149
166
  if (s >= ARRIVAL_SUSPICION) {
167
+ // What was actually on screen, not just that it was the wrong thing.
168
+ //
169
+ // This check has fired twice on CI and both times the report was a
170
+ // similarity score and two screen names, which is enough to know the
171
+ // run is void and not enough to know why. The tour asserts arrival with
172
+ // a `waitFor` before this ever runs, so a failure here means the
173
+ // `waitFor` *passed* on a screen that was not the destination — and the
174
+ // labels are the only thing that can say what that screen was.
175
+ const labels = (id.entry?.targets ?? [])
176
+ .map((t) => t.label).filter(Boolean).slice(0, 12).map((l) => l.slice(0, 24));
150
177
  arrivalFailures.push(
151
- `${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`);
178
+ `${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
179
+ + `\n sensors: ${(id.entry?.sources ?? []).join('+') || 'none'}`
180
+ + `, settled: ${id.settled}, frame ${Math.round(Date.now() - (id.state?.capturedAt ?? Date.now()))}ms old`
181
+ + `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`);
152
182
  }
153
183
  }
154
184
  readings.push({
@@ -199,6 +229,12 @@ if (arrivalFailures.length) {
199
229
  console.error('measure the tour rather than the fingerprint. Fix the tour and re-run.');
200
230
  console.error('A `settle` step defaults to mode "stable", which returns instantly in the');
201
231
  console.error('moment before an animation begins — action steps already settle on their own.');
232
+ console.error('');
233
+ console.error('The tour asserts arrival with a `waitFor` before any reading is taken, so a');
234
+ console.error('failure here means that wait PASSED on a screen that was not the destination.');
235
+ console.error('Read the labels above before changing the tour: either the wait matched');
236
+ console.error('something it should not have, or the reading came off a frame older than the');
237
+ console.error('navigation — and those have opposite fixes.');
202
238
  process.exit(1);
203
239
  }
204
240
 
@@ -339,7 +375,16 @@ console.log(`${thresholdInGap ? 'ok ' : 'WARN'} the threshold ${thresholdInGap
339
375
  const worstSame = [...same].sort((a, b) => a.similarity - b.similarity)[0];
340
376
  const worstDifferent = [...different].sort((a, b) => b.similarity - a.similarity)[0];
341
377
  if (worstSame) {
342
- console.log(`\nweakest same-screen pair: ${worstSame.a.name} r${worstSame.a.round} vs r${worstSame.b.round} = ${f(worstSame.similarity)} (${worstSame.a.count} vs ${worstSame.b.count} tokens)`);
378
+ // Say whether either side was read off a screen that was still moving.
379
+ //
380
+ // The counts have always been printed and the settle flag never was, so a red
381
+ // run showing `11 vs 6 tokens` left it open whether the fingerprint had
382
+ // drifted or one reading had simply been taken too early. It is printed per
383
+ // reading above, thirty lines away and on a different row; here it is beside
384
+ // the number it explains.
385
+ const moving = [worstSame.a, worstSame.b].filter((r) => r.settled === false).length;
386
+ const movingNote = moving ? `, ${moving === 2 ? 'both readings' : 'one reading'} taken on a screen that never settled` : '';
387
+ console.log(`\nweakest same-screen pair: ${worstSame.a.name} r${worstSame.a.round} vs r${worstSame.b.round} = ${f(worstSame.similarity)} (${worstSame.a.count} vs ${worstSame.b.count} tokens${movingNote})`);
343
388
  }
344
389
  if (worstDifferent) {
345
390
  console.log(`closest different-screen pair: ${worstDifferent.a.name} vs ${worstDifferent.b.name} = ${f(worstDifferent.similarity)}`);
@@ -0,0 +1,206 @@
1
+ #!/usr/bin/env node
2
+ // Ask several supervisors the same questions.
3
+ //
4
+ // node scripts/replay-rulings.mjs --device=<udid> --arms=apple,ollama:qwen3:8b,ollama:qwen3:14b
5
+ //
6
+ // The owner's call, 2026-09-11: settle the capacity question with numbers
7
+ // rather than speculation. This is how, and the shape matters more than the
8
+ // result.
9
+ //
10
+ // **Why replay rather than re-drive.** Collecting a population per arm means
11
+ // driving the simulator once per arm — about half an hour each — and, worse, it
12
+ // puts the *device's* variance inside a comparison that is supposed to be about
13
+ // the judges. A list that happened to arrive faster on one pass than another is
14
+ // not a fact about a model. So one device pass records the situations (see
15
+ // `situation` in metrics.recordSupervision) and every arm answers the identical
16
+ // set.
17
+ //
18
+ // **What replay cannot measure**, stated here rather than discovered later: the
19
+ // `outcome` column. Whether acting on a ruling actually recovered the flow is a
20
+ // fact about the device at that moment, and it belongs to the arm that was
21
+ // live. Replay scores *decisions*. The distinction is load-bearing — the live
22
+ // population scored 91% by decision and 69% by outcome, and almost all of that
23
+ // gap was one fixture that took the right word eight times and recovered once.
24
+ import * as metrics from '../src/metrics.js';
25
+ import * as ollama from '../src/ollama.js';
26
+ import * as supervisor from '../src/supervisor.js';
27
+ import { resolveDevice } from '../src/platform/index.js';
28
+
29
+ const arg = (n, d) => {
30
+ const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
31
+ return hit ? hit.slice(n.length + 3) : d;
32
+ };
33
+
34
+ /** The fixtures and the word each situation actually calls for. Same table as the scorer. */
35
+ const CORRECT = {
36
+ 'the list is still loading; its rows arrive shortly after launch': 'wait',
37
+ 'the detail screen is still fetching; its text arrives shortly': 'wait',
38
+ 'the list arrives in waves and this row is in the last one': 'wait',
39
+ 'Review is blocked until Species is filled in, and it is empty': 'stop',
40
+ 'the first submit always fails and the second works, so waiting cannot help': 'stop',
41
+ };
42
+
43
+ /** `retry` and `wait` differ only in how long they settle, so both satisfy a wait. */
44
+ const satisfies = (d, want) => (want === 'wait' ? d === 'wait' || d === 'retry' : d === want);
45
+
46
+ const dev = await resolveDevice(arg('device'));
47
+ const arms = String(arg('arms', 'apple')).split(',').map((a) => a.trim()).filter(Boolean);
48
+ const lastN = Number(arg('last', 0));
49
+
50
+ const all = metrics.readSupervisions(dev.udid)
51
+ .filter((r) => CORRECT[r.expect] && r.situation?.step && r.situation?.failure);
52
+ const rows = lastN > 0 ? all.slice(-lastN) : all;
53
+
54
+ if (rows.length < 8) {
55
+ console.error(`only ${rows.length} replayable ruling(s) — need situations recorded, which`);
56
+ console.error('means a population collected after the situation field was added.');
57
+ console.error('Run: SIMFRAME_SUPERVISOR=apple node scripts/collect-rulings.mjs --device=<udid>');
58
+ process.exit(2);
59
+ }
60
+
61
+ const want = (r) => CORRECT[r.expect];
62
+ const counts = rows.reduce((a, r) => ({ ...a, [want(r)]: (a[want(r)] ?? 0) + 1 }), {});
63
+ const commonest = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
64
+ const baseline = commonest[1] / rows.length;
65
+ const pct = (x) => `${(100 * x).toFixed(0)}%`;
66
+
67
+ console.log(`${rows.length} situations from ${dev.name}`);
68
+ console.log(`balance: ${JSON.stringify(counts)}`);
69
+ console.log(`majority-class baseline: always "${commonest[0]}" scores ${pct(baseline)}`);
70
+ if (baseline > 0.65) {
71
+ console.log('\nSKEWED — any accuracy below is mostly a fact about the fixture set.');
72
+ }
73
+ // The free comparison, computed here so it is in the same table as the models
74
+ // rather than in a different report. It is not an arm; it is the thing every
75
+ // arm has to beat to be worth its latency.
76
+ const STILL_MS_THRESHOLD = 3000;
77
+ console.log(`\nthe brief every model arm gets is ${ollama.readBrief().length} characters, read from native/supervise.swift`);
78
+
79
+ const withAbstain = process.argv.includes('--abstain');
80
+
81
+ const results = [];
82
+ for (const arm of arms) {
83
+ const judged = [];
84
+ let unanswered = 0;
85
+ // Weights first, so the first situation is not timing a disk read.
86
+ if (arm.startsWith('ollama')) await ollama.preload(ollama.parseTarget(arm));
87
+ process.stdout.write(`\nasking ${arm} …`);
88
+ for (const r of rows) {
89
+ const detail = {};
90
+ const ruling = await supervisor.judge({
91
+ ...r.situation,
92
+ mayAbstain: withAbstain,
93
+ options: { supervisor: arm },
94
+ // Wide on purpose. The shipped budget is 2.5s and a 14B will exceed it;
95
+ // capping here would score the larger model on *latency* while calling it
96
+ // accuracy, and latency is reported separately below where it can be read
97
+ // for what it is.
98
+ timeoutMs: 30_000,
99
+ detail,
100
+ });
101
+ if (!ruling) unanswered += 1;
102
+ judged.push({ r, ruling, detail, abstained: detail.kind === 'abstained' });
103
+ process.stdout.write('.');
104
+ }
105
+ const answered = judged.filter((j) => j.ruling);
106
+ const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
107
+ const lat = answered.map((j) => j.ruling.ms).filter(Number.isFinite).sort((a, b) => a - b);
108
+ results.push({
109
+ arm,
110
+ judged,
111
+ n: rows.length,
112
+ unanswered,
113
+ // Scored over every situation, not only the answered ones. A judge that
114
+ // declines half the questions and is right about the rest is not an 100%
115
+ // judge, and scoring only its answers would say it was.
116
+ accuracy: right / rows.length,
117
+ medianMs: lat.length ? lat[lat.length >> 1] : null,
118
+ errors: answered
119
+ .filter((j) => !satisfies(j.ruling.decision, want(j.r)))
120
+ .map((j) => `said "${j.ruling.decision}" where "${want(j.r)}" was right — ${j.r.expect.slice(0, 48)}`),
121
+ });
122
+ process.stdout.write(' done\n');
123
+ }
124
+
125
+ const stillRule = rows.filter((r) => {
126
+ const d = (r.situation.stillMs ?? 0) > STILL_MS_THRESHOLD ? 'stop' : 'wait';
127
+ return satisfies(d, want(r));
128
+ }).length / rows.length;
129
+
130
+ console.log(`\n${'arm'.padEnd(26)} ${'accuracy'.padStart(9)} ${'median'.padStart(9)} ${'no answer'.padStart(10)}`);
131
+ console.log('-'.repeat(58));
132
+ console.log(`${`always "${commonest[0]}"`.padEnd(26)} ${pct(baseline).padStart(9)} ${'—'.padStart(9)} ${'—'.padStart(10)}`);
133
+ console.log(`${`stillMs > ${STILL_MS_THRESHOLD}ms`.padEnd(26)} ${pct(stillRule).padStart(9)} ${'0ms'.padStart(9)} ${'—'.padStart(10)}`);
134
+ for (const r of results) {
135
+ console.log(`${r.arm.padEnd(26)} ${pct(r.accuracy).padStart(9)} ${`${r.medianMs ?? '—'}ms`.padStart(9)} ${`${r.unanswered}`.padStart(10)}`);
136
+ }
137
+
138
+ if (withAbstain) {
139
+ console.log(`\n--- item 100: the fourth word ---`);
140
+ console.log('The question is not whether it abstains. It is whether it abstains on');
141
+ console.log('the ones it would have got WRONG, or at random. A judge that declines');
142
+ console.log('uniformly has added latency and a round trip and bought nothing.\n');
143
+ console.log(`${'arm'.padEnd(22)} ${'answered'.padStart(9)} ${'of those'.padStart(9)} ${'abstained'.padStart(10)} ${'escalations'.padStart(12)}`);
144
+ console.log('-'.repeat(66));
145
+ for (const r of results) {
146
+ const answered = r.judged.filter((j) => j.ruling);
147
+ const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
148
+ const abstained = r.judged.filter((j) => j.abstained).length;
149
+ console.log(`${r.arm.padEnd(22)} ${`${answered.length}/${rows.length}`.padStart(9)} ${pct(right / (answered.length || 1)).padStart(9)} ${`${abstained}`.padStart(10)} ${pct(abstained / rows.length).padStart(12)}`);
150
+ }
151
+ console.log('\n`of those` is accuracy on the questions it chose to answer. If the fourth');
152
+ console.log('word is working, that number is higher than the three-word accuracy above');
153
+ console.log('by more than the abstention rate would give by chance.');
154
+ }
155
+
156
+ console.log('\nerrors, by arm:');
157
+ for (const r of results) {
158
+ console.log(` ${r.arm}`);
159
+ if (!r.errors.length) console.log(' none');
160
+ const seen = new Map();
161
+ for (const e of r.errors) seen.set(e, (seen.get(e) ?? 0) + 1);
162
+ for (const [e, n] of [...seen].sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${e}`);
163
+ }
164
+
165
+ // --- the cascade: threshold -> local model -> Claude -----------------------
166
+ //
167
+ // The owner's proposal, and the numbers are the only way to say whether it
168
+ // helps: answer with the free rule where it is confident, fall through to the
169
+ // on-device model where it is not, and only then pay a round trip.
170
+ //
171
+ // **A cascade needs a tier that can decline, and neither tier has one.** The
172
+ // threshold is a comparison — it always answers. The supervisor's vocabulary is
173
+ // three words and none of them is "I don't know". So the interesting number is
174
+ // not "does a cascade help" but "how much would an abstain token be worth", and
175
+ // that is item 100. This measures it directly, by letting the rule abstain in a
176
+ // band around its own threshold and handing those to the next tier.
177
+ const armDecisions = new Map(results.map((r) => [r.arm, r.judged]));
178
+ console.log('\n--- cascade: the rule answers, the model covers where it abstains ---');
179
+ console.log(`${'abstain band'.padEnd(22)} ${'rule'.padStart(6)} ${'->model'.padStart(8)} ${'cascade'.padStart(9)} ${'model calls'.padStart(12)}`);
180
+ for (const band of [0, 250, 500, 750, 1000, 1500]) {
181
+ const lo = STILL_MS_THRESHOLD - band;
182
+ const hi = STILL_MS_THRESHOLD + band;
183
+ for (const arm of arms) {
184
+ const judged = armDecisions.get(arm) ?? [];
185
+ let right = 0;
186
+ let escalated = 0;
187
+ for (const [i, r] of rows.entries()) {
188
+ const still = r.situation.stillMs ?? 0;
189
+ if (band > 0 && still >= lo && still <= hi) {
190
+ escalated += 1;
191
+ const ruling = judged[i]?.ruling;
192
+ if (ruling && satisfies(ruling.decision, want(r))) right += 1;
193
+ continue;
194
+ }
195
+ if (satisfies(still > STILL_MS_THRESHOLD ? 'stop' : 'wait', want(r))) right += 1;
196
+ }
197
+ console.log(`${`+/-${band}ms -> ${arm}`.padEnd(22)} ${pct(stillRule).padStart(6)} ${`${escalated}`.padStart(8)} ${pct(right / rows.length).padStart(9)} ${`${escalated}/${rows.length}`.padStart(12)}`);
198
+ }
199
+ if (band === 0) console.log(' (band 0 = no abstention, the rule alone — every row below adds a tier)');
200
+ }
201
+
202
+ console.log('\nThis scores DECISIONS on identical inputs. It cannot score outcomes —');
203
+ console.log('whether acting on a ruling recovered the flow is a fact about the device at');
204
+ console.log('that moment, and belongs to whichever arm was live. See docs/EXPERIMENTS.md.');
205
+
206
+ supervisor.close();
@@ -42,7 +42,25 @@ const all = metrics.readSupervisions(dev.udid).filter((r) => CORRECT[r.expect]);
42
42
  // test was 16/16. Mixing a known-bad population into the denominator is the
43
43
  // same error as before wearing a different hat.
44
44
  const lastN = Number(arg('last', 0));
45
- const rows = lastN > 0 ? all.slice(-lastN) : all;
45
+ // Which judge. Every arm of the capacity comparison writes to one log, so
46
+ // scoring without this would average a 3B, an 8B and a 14B into a single
47
+ // meaningless number — and it would look like a result.
48
+ const arm = arg('arm', null);
49
+ const armOf = (r) => r.supervisor ?? 'unrecorded';
50
+ const scoped = arm ? all.filter((r) => armOf(r) === arm) : all;
51
+ const rows = lastN > 0 ? scoped.slice(-lastN) : scoped;
52
+
53
+ // A population spanning several arms is not one population. Say so, and say
54
+ // what to pass, rather than printing an average of things that were never
55
+ // comparable.
56
+ const arms = [...new Set(rows.map(armOf))];
57
+ if (arms.length > 1) {
58
+ console.log(`this log holds rulings from ${arms.length} supervisor arms:`);
59
+ for (const a of arms) console.log(` ${String(rows.filter((r) => armOf(r) === a).length).padStart(4)} ${a}`);
60
+ console.log('\nScore one at a time — `--arm=<name>` — because averaging them is not a result.');
61
+ process.exit(2);
62
+ }
63
+ if (rows.length) console.log(`arm: ${arms[0]}\n`);
46
64
  if (rows.length < 8) {
47
65
  console.error(`only ${rows.length} labelled ruling(s) — run scripts/collect-rulings.mjs first`);
48
66
  process.exit(2);