simframe 0.12.1 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +163 -9
- package/native/supervise.swift +63 -4
- package/package.json +1 -1
- package/scripts/article-md.mjs +185 -0
- package/scripts/ci-integration-local.sh +22 -1
- package/scripts/ci-memory.mjs +98 -15
- package/scripts/collect-rulings.mjs +14 -0
- package/scripts/eval-fingerprint.mjs +48 -3
- package/scripts/replay-rulings.mjs +206 -0
- package/scripts/score-rulings.mjs +19 -1
- package/src/actions.js +158 -18
- package/src/analyze.js +56 -0
- package/src/cli.js +215 -18
- package/src/fingerprint.js +10 -1
- package/src/index.js +337 -12
- package/src/input.js +4 -0
- package/src/mcp.js +7 -1
- package/src/metrics.js +68 -7
- package/src/navigate.js +28 -2
- package/src/ollama.js +232 -0
- package/src/platform/ios.js +30 -5
- package/src/refs.js +12 -1
- package/src/regions.js +54 -0
- package/src/screenmap.js +161 -6
- package/src/store.js +53 -0
- package/src/supervisor.js +108 -5
- package/src/view.js +84 -9
package/scripts/ci-memory.mjs
CHANGED
|
@@ -50,6 +50,9 @@ let failures = 0;
|
|
|
50
50
|
*/
|
|
51
51
|
let deviceDied = null;
|
|
52
52
|
|
|
53
|
+
/** EX_TEMPFAIL: the device died under the checks, so nothing was tested. */
|
|
54
|
+
const DEVICE_DIED_EXIT = 75;
|
|
55
|
+
|
|
53
56
|
function check(ok, label, detail = '') {
|
|
54
57
|
if (!ok) failures += 1;
|
|
55
58
|
console.log(`${ok ? 'ok ' : 'FAIL'} ${label}${detail ? ` — ${detail}` : ''}`);
|
|
@@ -171,9 +174,21 @@ async function jsonRetry(args, opts, attempts = 3) {
|
|
|
171
174
|
console.error(` ${String(last.message).split('\n')[0]}`);
|
|
172
175
|
console.error('\nEverything after this point would be testing a dead simulator, so the run');
|
|
173
176
|
console.error('stops here. This is not a memory-layer failure — it is the device-state');
|
|
174
|
-
console.error('problem in docs/DEFERRED.md. A device restart is the only known cure
|
|
175
|
-
console.error('
|
|
176
|
-
|
|
177
|
+
console.error('problem in docs/DEFERRED.md (126). A device restart is the only known cure,');
|
|
178
|
+
console.error('and `simframe revive` is that restart.');
|
|
179
|
+
// Exit 75, not 1, and the distinction is the whole point of this file.
|
|
180
|
+
//
|
|
181
|
+
// "A check about the memory layer failed" and "the simulator died under the
|
|
182
|
+
// checks" are different conditions with different responses, and for two CI
|
|
183
|
+
// rounds they were one exit code — so a caller could only retry everything
|
|
184
|
+
// or retry nothing. 75 is EX_TEMPFAIL, which is exactly what this is: the
|
|
185
|
+
// subject under test was never reached.
|
|
186
|
+
//
|
|
187
|
+
// The caller reviving and running again is not papering over a product bug.
|
|
188
|
+
// The wedge is a documented CoreSimulator condition, `frame --fresh` names
|
|
189
|
+
// it, and `revive` is the cure this project ships for it — CI simply had no
|
|
190
|
+
// way to say "use it".
|
|
191
|
+
process.exit(DEVICE_DIED_EXIT);
|
|
177
192
|
}
|
|
178
193
|
throw last;
|
|
179
194
|
}
|
|
@@ -244,7 +259,36 @@ if (failures) {
|
|
|
244
259
|
|
|
245
260
|
console.log('\n--- the screen map ---');
|
|
246
261
|
await jsonRetry(['do', LOOP], { allowFail: true });
|
|
247
|
-
const map = await
|
|
262
|
+
const map = await readableMap();
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* A map that is empty *and* says a sensor failed is a blink, not a result.
|
|
266
|
+
*
|
|
267
|
+
* `jsonRetry` retries a command that throws, and this one did not throw: `ui`
|
|
268
|
+
* returned 200 with zero elements and a degraded note, which is the shape of
|
|
269
|
+
* failure this whole repo keeps writing rules about. The run that made this
|
|
270
|
+
* necessary had the accessibility read time out once, report an empty map, and
|
|
271
|
+
* then resolve a ref correctly sixty seconds later on the same device — so the
|
|
272
|
+
* device was fine and the check had caught one bad read.
|
|
273
|
+
*
|
|
274
|
+
* Deliberately narrow. An empty map with **no** degraded sensor is a real
|
|
275
|
+
* answer — that is a blank screen and the check should fail on it. Only an
|
|
276
|
+
* empty map that admits a layer did not answer is worth asking again, and after
|
|
277
|
+
* three attempts it fails with what the sensor said, which is the diagnosis
|
|
278
|
+
* either way.
|
|
279
|
+
*/
|
|
280
|
+
async function readableMap(attempts = 3) {
|
|
281
|
+
let last;
|
|
282
|
+
for (let i = 0; i < attempts; i += 1) {
|
|
283
|
+
last = await jsonRetry(['ui']);
|
|
284
|
+
if (last.elements?.length || !(last.degraded ?? []).length) return last;
|
|
285
|
+
if (i < attempts - 1) {
|
|
286
|
+
console.log(` (the map came back empty and a sensor said why — retrying \`ui\`: ${(last.degraded ?? []).join('; ')})`);
|
|
287
|
+
await new Promise((r) => setTimeout(r, 2000));
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
return last;
|
|
291
|
+
}
|
|
248
292
|
|
|
249
293
|
// Report what actually answered rather than asserting the runner's situation.
|
|
250
294
|
// This line used to read "with no accessibility tree available" unconditionally,
|
|
@@ -263,14 +307,38 @@ check(Number.isFinite(map.points?.width) && Number.isFinite(map.points?.height),
|
|
|
263
307
|
'the map knows the screen size in points', `${map.points?.width}x${map.points?.height}pt`);
|
|
264
308
|
|
|
265
309
|
const refs = (map.elements ?? []).map((e) => e.ref);
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
310
|
+
// Three checks over a collection, and `every`/`!some` are all true of an empty
|
|
311
|
+
// one. On a CI run whose map came back with **0 elements** they printed
|
|
312
|
+
// `ok refs are numbered 1..n with no gaps — #1..#0` and two more like it:
|
|
313
|
+
// three lines of reassurance about nothing, directly under the failure that
|
|
314
|
+
// said the map was empty.
|
|
315
|
+
//
|
|
316
|
+
// That is the exact defect three field reports spent a day describing — a
|
|
317
|
+
// confident statement that verified nothing — and the harness was doing it to
|
|
318
|
+
// itself in the same output. Untested is not passed.
|
|
319
|
+
if (!map.elements?.length) {
|
|
320
|
+
for (const label of [
|
|
321
|
+
'refs are numbered 1..n with no gaps',
|
|
322
|
+
'every element has a tap point on the screen',
|
|
323
|
+
'the status bar is not offered as something to tap',
|
|
324
|
+
]) skip(label, 'the map was empty, so there was nothing to check');
|
|
325
|
+
// And say what the device was showing, because an empty map on a live device
|
|
326
|
+
// is the shape this project has chased under four different symptoms. The
|
|
327
|
+
// liveness note carries the ignored-gesture check; `stableForMs` and the
|
|
328
|
+
// frame age are the two numbers whose disagreement names a dead surface.
|
|
329
|
+
const st = await jsonRetry(['state'], { allowFail: true });
|
|
330
|
+
console.log(` the device at that moment: frame #${st?.seq ?? '?'}, `
|
|
331
|
+
+ `${st?.stableForMs ?? '?'}ms still, ${st?.live?.note ?? 'liveness reported nothing'}`);
|
|
332
|
+
} else {
|
|
333
|
+
check(refs.every((r, i) => r === i + 1),
|
|
334
|
+
'refs are numbered 1..n with no gaps', `#1..#${refs.length}`);
|
|
335
|
+
check(map.elements.every((e) =>
|
|
336
|
+
Number.isInteger(e.x) && Number.isInteger(e.y)
|
|
337
|
+
&& e.y >= 0 && e.y <= map.points.height && e.x >= 0 && e.x <= map.points.width),
|
|
338
|
+
'every element has a tap point on the screen');
|
|
339
|
+
check(!map.elements.some((e) => e.region === 'status-bar'),
|
|
340
|
+
'the status bar is not offered as something to tap');
|
|
341
|
+
}
|
|
274
342
|
|
|
275
343
|
console.log('\n--- element refs ---');
|
|
276
344
|
// Re-read on a screen we chose, rather than on whatever the device happened to
|
|
@@ -296,7 +364,8 @@ if (first) {
|
|
|
296
364
|
// hashes only have to agree with that, and they only get a say when they are
|
|
297
365
|
// informative enough to have one.
|
|
298
366
|
const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
|
|
299
|
-
|
|
367
|
+
const launched = ran(left);
|
|
368
|
+
if (!launched) skip('the screen actually changed before testing the stale ref',
|
|
300
369
|
'the second app never launched, so there was no screen change to test against');
|
|
301
370
|
else check(moved, 'the screen actually changed before testing the stale ref',
|
|
302
371
|
`${before.slice(0, 10)} -> ${after.slice(0, 10)}`
|
|
@@ -305,7 +374,16 @@ if (first) {
|
|
|
305
374
|
// reports "the stale-ref guard failed" for a device that never left the
|
|
306
375
|
// screen, which is a false accusation against the one layer this file exists
|
|
307
376
|
// to defend — and it is how this check has failed twice.
|
|
308
|
-
|
|
377
|
+
// A skip has to propagate. The precondition above reported NOT TESTED and this
|
|
378
|
+
// check ran anyway and failed — which is the harness doing to itself, one line
|
|
379
|
+
// later, exactly what `skip` was written to stop it doing. `moved` is true
|
|
380
|
+
// when a hash is too degenerate to have a say, and that is right for "did the
|
|
381
|
+
// screen change" and wrong as a licence to run a check whose setup is known
|
|
382
|
+
// not to have happened.
|
|
383
|
+
if (!launched) {
|
|
384
|
+
skip('a ref numbered on another screen refuses instead of tapping those coordinates',
|
|
385
|
+
'we never reached another screen, so there was nothing to refuse from');
|
|
386
|
+
} else if (moved) {
|
|
309
387
|
// This matched on prose twice and went red twice, both times for a refusal
|
|
310
388
|
// that was correct and better worded than the alternation knew — most
|
|
311
389
|
// recently `"Welcome to Reminders" is not on this screen`, which refuses
|
|
@@ -491,7 +569,12 @@ const walked = await jsonRetry(['goto', target.hash], { allowFail: true });
|
|
|
491
569
|
// neither walking there nor naming why it cannot, so this check failed with an
|
|
492
570
|
// empty detail — the reason was `undefined` — and the check was right to fail.
|
|
493
571
|
// Now there is a name for it, and the list has to know the name.
|
|
494
|
-
|
|
572
|
+
// `route-halted` and `arrived-elsewhere` are new: a walk that ran and did not
|
|
573
|
+
// land used to return `{ok: false}` with no reason at all, which failed this
|
|
574
|
+
// very check with an empty detail. It was the one outcome here nobody had
|
|
575
|
+
// named, and the check found it.
|
|
576
|
+
const outcomes = ['no-route', 'unreplayable-edge', 'ambiguous', 'unknown-screen', 'no-identity',
|
|
577
|
+
'route-halted', 'arrived-elsewhere'];
|
|
495
578
|
check(walked.ok === true || outcomes.includes(walked.reason),
|
|
496
579
|
'and asked for a screen it knows, it either walks there or names why it cannot',
|
|
497
580
|
walked.ok ? (walked.already ? 'already there' : `walked ${walked.ranSteps} step(s)`) : walked.reason);
|
|
@@ -310,3 +310,17 @@ if (unattributed) console.log(`\n${unattributed} ruling(s) came from a step this
|
|
|
310
310
|
}
|
|
311
311
|
}
|
|
312
312
|
console.log(`\nthe log now holds ${all.length} ruling(s) — simframe supervisions --device=${dev.udid}`);
|
|
313
|
+
|
|
314
|
+
// Close the helper, or this script never exits.
|
|
315
|
+
//
|
|
316
|
+
// The supervisor keeps a warm child process and deliberately does not unref its
|
|
317
|
+
// stdout — unreferencing it once unreferenced the pipe every request waits on,
|
|
318
|
+
// and the process then exited silently mid-await. The consequence nobody had
|
|
319
|
+
// noticed is at *this* end: the script finishes its work, prints this line, and
|
|
320
|
+
// then sits at 0% CPU forever with the helper idle at `readLine`, because the
|
|
321
|
+
// live child is holding the event loop open.
|
|
322
|
+
//
|
|
323
|
+
// That is the whole story of three "orphaned" collect-rulings processes killed
|
|
324
|
+
// by PID the night before, which were attributed to a task runner not killing
|
|
325
|
+
// its children. They had each finished their work and could not leave.
|
|
326
|
+
supervisor.close();
|
|
@@ -129,7 +129,24 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
129
129
|
if (screen.steps?.length) {
|
|
130
130
|
// Verification is off: this measures fingerprints, and a wrong-turn
|
|
131
131
|
// verdict computed from the very tokens under test would be circular.
|
|
132
|
-
|
|
132
|
+
//
|
|
133
|
+
// The tour asserts arrival rather than sleeping through it, so a step
|
|
134
|
+
// here can now fail — `openUrl` returns NSPOSIXErrorDomain 60 on a loaded
|
|
135
|
+
// hosted runner, and a page that never renders no longer passes silently
|
|
136
|
+
// as a six-token reading. That is the trade this harness wants: a loud
|
|
137
|
+
// failure naming the screen, over a quiet one that shows up thirty lines
|
|
138
|
+
// later as a threshold with no clearance. Everything read so far is
|
|
139
|
+
// written out first, because a failing run is the one whose evidence
|
|
140
|
+
// matters.
|
|
141
|
+
try {
|
|
142
|
+
await actions.runScript(device, { steps: screen.steps, verify: false });
|
|
143
|
+
} catch (err) {
|
|
144
|
+
save({ abandonedAt: { screen: screen.name, round, error: err.message } });
|
|
145
|
+
console.error(`\nFAIL round ${round}, "${screen.name}" never arrived: ${err.message}`);
|
|
146
|
+
console.error('The tour waits for something each screen actually shows. This is that wait');
|
|
147
|
+
console.error('giving up — not a fingerprint result. Check the app, the network, or the runner.');
|
|
148
|
+
process.exit(1);
|
|
149
|
+
}
|
|
133
150
|
}
|
|
134
151
|
const id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
135
152
|
// Did we actually arrive? Two differently-named screens reading the same
|
|
@@ -147,8 +164,21 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
147
164
|
// means the navigation did not happen.
|
|
148
165
|
const s = fingerprint.similarity(previous.tokens, id.tokens ?? []);
|
|
149
166
|
if (s >= ARRIVAL_SUSPICION) {
|
|
167
|
+
// What was actually on screen, not just that it was the wrong thing.
|
|
168
|
+
//
|
|
169
|
+
// This check has fired twice on CI and both times the report was a
|
|
170
|
+
// similarity score and two screen names, which is enough to know the
|
|
171
|
+
// run is void and not enough to know why. The tour asserts arrival with
|
|
172
|
+
// a `waitFor` before this ever runs, so a failure here means the
|
|
173
|
+
// `waitFor` *passed* on a screen that was not the destination — and the
|
|
174
|
+
// labels are the only thing that can say what that screen was.
|
|
175
|
+
const labels = (id.entry?.targets ?? [])
|
|
176
|
+
.map((t) => t.label).filter(Boolean).slice(0, 12).map((l) => l.slice(0, 24));
|
|
150
177
|
arrivalFailures.push(
|
|
151
|
-
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
|
|
178
|
+
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
|
|
179
|
+
+ `\n sensors: ${(id.entry?.sources ?? []).join('+') || 'none'}`
|
|
180
|
+
+ `, settled: ${id.settled}, frame ${Math.round(Date.now() - (id.state?.capturedAt ?? Date.now()))}ms old`
|
|
181
|
+
+ `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`);
|
|
152
182
|
}
|
|
153
183
|
}
|
|
154
184
|
readings.push({
|
|
@@ -199,6 +229,12 @@ if (arrivalFailures.length) {
|
|
|
199
229
|
console.error('measure the tour rather than the fingerprint. Fix the tour and re-run.');
|
|
200
230
|
console.error('A `settle` step defaults to mode "stable", which returns instantly in the');
|
|
201
231
|
console.error('moment before an animation begins — action steps already settle on their own.');
|
|
232
|
+
console.error('');
|
|
233
|
+
console.error('The tour asserts arrival with a `waitFor` before any reading is taken, so a');
|
|
234
|
+
console.error('failure here means that wait PASSED on a screen that was not the destination.');
|
|
235
|
+
console.error('Read the labels above before changing the tour: either the wait matched');
|
|
236
|
+
console.error('something it should not have, or the reading came off a frame older than the');
|
|
237
|
+
console.error('navigation — and those have opposite fixes.');
|
|
202
238
|
process.exit(1);
|
|
203
239
|
}
|
|
204
240
|
|
|
@@ -339,7 +375,16 @@ console.log(`${thresholdInGap ? 'ok ' : 'WARN'} the threshold ${thresholdInGap
|
|
|
339
375
|
const worstSame = [...same].sort((a, b) => a.similarity - b.similarity)[0];
|
|
340
376
|
const worstDifferent = [...different].sort((a, b) => b.similarity - a.similarity)[0];
|
|
341
377
|
if (worstSame) {
|
|
342
|
-
|
|
378
|
+
// Say whether either side was read off a screen that was still moving.
|
|
379
|
+
//
|
|
380
|
+
// The counts have always been printed and the settle flag never was, so a red
|
|
381
|
+
// run showing `11 vs 6 tokens` left it open whether the fingerprint had
|
|
382
|
+
// drifted or one reading had simply been taken too early. It is printed per
|
|
383
|
+
// reading above, thirty lines away and on a different row; here it is beside
|
|
384
|
+
// the number it explains.
|
|
385
|
+
const moving = [worstSame.a, worstSame.b].filter((r) => r.settled === false).length;
|
|
386
|
+
const movingNote = moving ? `, ${moving === 2 ? 'both readings' : 'one reading'} taken on a screen that never settled` : '';
|
|
387
|
+
console.log(`\nweakest same-screen pair: ${worstSame.a.name} r${worstSame.a.round} vs r${worstSame.b.round} = ${f(worstSame.similarity)} (${worstSame.a.count} vs ${worstSame.b.count} tokens${movingNote})`);
|
|
343
388
|
}
|
|
344
389
|
if (worstDifferent) {
|
|
345
390
|
console.log(`closest different-screen pair: ${worstDifferent.a.name} vs ${worstDifferent.b.name} = ${f(worstDifferent.similarity)}`);
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Ask several supervisors the same questions.
|
|
3
|
+
//
|
|
4
|
+
// node scripts/replay-rulings.mjs --device=<udid> --arms=apple,ollama:qwen3:8b,ollama:qwen3:14b
|
|
5
|
+
//
|
|
6
|
+
// The owner's call, 2026-09-11: settle the capacity question with numbers
|
|
7
|
+
// rather than speculation. This is how, and the shape matters more than the
|
|
8
|
+
// result.
|
|
9
|
+
//
|
|
10
|
+
// **Why replay rather than re-drive.** Collecting a population per arm means
|
|
11
|
+
// driving the simulator once per arm — about half an hour each — and, worse, it
|
|
12
|
+
// puts the *device's* variance inside a comparison that is supposed to be about
|
|
13
|
+
// the judges. A list that happened to arrive faster on one pass than another is
|
|
14
|
+
// not a fact about a model. So one device pass records the situations (see
|
|
15
|
+
// `situation` in metrics.recordSupervision) and every arm answers the identical
|
|
16
|
+
// set.
|
|
17
|
+
//
|
|
18
|
+
// **What replay cannot measure**, stated here rather than discovered later: the
|
|
19
|
+
// `outcome` column. Whether acting on a ruling actually recovered the flow is a
|
|
20
|
+
// fact about the device at that moment, and it belongs to the arm that was
|
|
21
|
+
// live. Replay scores *decisions*. The distinction is load-bearing — the live
|
|
22
|
+
// population scored 91% by decision and 69% by outcome, and almost all of that
|
|
23
|
+
// gap was one fixture that took the right word eight times and recovered once.
|
|
24
|
+
import * as metrics from '../src/metrics.js';
|
|
25
|
+
import * as ollama from '../src/ollama.js';
|
|
26
|
+
import * as supervisor from '../src/supervisor.js';
|
|
27
|
+
import { resolveDevice } from '../src/platform/index.js';
|
|
28
|
+
|
|
29
|
+
const arg = (n, d) => {
|
|
30
|
+
const hit = process.argv.find((a) => a.startsWith(`--${n}=`));
|
|
31
|
+
return hit ? hit.slice(n.length + 3) : d;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
/** The fixtures and the word each situation actually calls for. Same table as the scorer. */
|
|
35
|
+
const CORRECT = {
|
|
36
|
+
'the list is still loading; its rows arrive shortly after launch': 'wait',
|
|
37
|
+
'the detail screen is still fetching; its text arrives shortly': 'wait',
|
|
38
|
+
'the list arrives in waves and this row is in the last one': 'wait',
|
|
39
|
+
'Review is blocked until Species is filled in, and it is empty': 'stop',
|
|
40
|
+
'the first submit always fails and the second works, so waiting cannot help': 'stop',
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
/** `retry` and `wait` differ only in how long they settle, so both satisfy a wait. */
|
|
44
|
+
const satisfies = (d, want) => (want === 'wait' ? d === 'wait' || d === 'retry' : d === want);
|
|
45
|
+
|
|
46
|
+
const dev = await resolveDevice(arg('device'));
|
|
47
|
+
const arms = String(arg('arms', 'apple')).split(',').map((a) => a.trim()).filter(Boolean);
|
|
48
|
+
const lastN = Number(arg('last', 0));
|
|
49
|
+
|
|
50
|
+
const all = metrics.readSupervisions(dev.udid)
|
|
51
|
+
.filter((r) => CORRECT[r.expect] && r.situation?.step && r.situation?.failure);
|
|
52
|
+
const rows = lastN > 0 ? all.slice(-lastN) : all;
|
|
53
|
+
|
|
54
|
+
if (rows.length < 8) {
|
|
55
|
+
console.error(`only ${rows.length} replayable ruling(s) — need situations recorded, which`);
|
|
56
|
+
console.error('means a population collected after the situation field was added.');
|
|
57
|
+
console.error('Run: SIMFRAME_SUPERVISOR=apple node scripts/collect-rulings.mjs --device=<udid>');
|
|
58
|
+
process.exit(2);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const want = (r) => CORRECT[r.expect];
|
|
62
|
+
const counts = rows.reduce((a, r) => ({ ...a, [want(r)]: (a[want(r)] ?? 0) + 1 }), {});
|
|
63
|
+
const commonest = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
|
|
64
|
+
const baseline = commonest[1] / rows.length;
|
|
65
|
+
const pct = (x) => `${(100 * x).toFixed(0)}%`;
|
|
66
|
+
|
|
67
|
+
console.log(`${rows.length} situations from ${dev.name}`);
|
|
68
|
+
console.log(`balance: ${JSON.stringify(counts)}`);
|
|
69
|
+
console.log(`majority-class baseline: always "${commonest[0]}" scores ${pct(baseline)}`);
|
|
70
|
+
if (baseline > 0.65) {
|
|
71
|
+
console.log('\nSKEWED — any accuracy below is mostly a fact about the fixture set.');
|
|
72
|
+
}
|
|
73
|
+
// The free comparison, computed here so it is in the same table as the models
|
|
74
|
+
// rather than in a different report. It is not an arm; it is the thing every
|
|
75
|
+
// arm has to beat to be worth its latency.
|
|
76
|
+
const STILL_MS_THRESHOLD = 3000;
|
|
77
|
+
console.log(`\nthe brief every model arm gets is ${ollama.readBrief().length} characters, read from native/supervise.swift`);
|
|
78
|
+
|
|
79
|
+
const withAbstain = process.argv.includes('--abstain');
|
|
80
|
+
|
|
81
|
+
const results = [];
|
|
82
|
+
for (const arm of arms) {
|
|
83
|
+
const judged = [];
|
|
84
|
+
let unanswered = 0;
|
|
85
|
+
// Weights first, so the first situation is not timing a disk read.
|
|
86
|
+
if (arm.startsWith('ollama')) await ollama.preload(ollama.parseTarget(arm));
|
|
87
|
+
process.stdout.write(`\nasking ${arm} …`);
|
|
88
|
+
for (const r of rows) {
|
|
89
|
+
const detail = {};
|
|
90
|
+
const ruling = await supervisor.judge({
|
|
91
|
+
...r.situation,
|
|
92
|
+
mayAbstain: withAbstain,
|
|
93
|
+
options: { supervisor: arm },
|
|
94
|
+
// Wide on purpose. The shipped budget is 2.5s and a 14B will exceed it;
|
|
95
|
+
// capping here would score the larger model on *latency* while calling it
|
|
96
|
+
// accuracy, and latency is reported separately below where it can be read
|
|
97
|
+
// for what it is.
|
|
98
|
+
timeoutMs: 30_000,
|
|
99
|
+
detail,
|
|
100
|
+
});
|
|
101
|
+
if (!ruling) unanswered += 1;
|
|
102
|
+
judged.push({ r, ruling, detail, abstained: detail.kind === 'abstained' });
|
|
103
|
+
process.stdout.write('.');
|
|
104
|
+
}
|
|
105
|
+
const answered = judged.filter((j) => j.ruling);
|
|
106
|
+
const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
|
|
107
|
+
const lat = answered.map((j) => j.ruling.ms).filter(Number.isFinite).sort((a, b) => a - b);
|
|
108
|
+
results.push({
|
|
109
|
+
arm,
|
|
110
|
+
judged,
|
|
111
|
+
n: rows.length,
|
|
112
|
+
unanswered,
|
|
113
|
+
// Scored over every situation, not only the answered ones. A judge that
|
|
114
|
+
// declines half the questions and is right about the rest is not an 100%
|
|
115
|
+
// judge, and scoring only its answers would say it was.
|
|
116
|
+
accuracy: right / rows.length,
|
|
117
|
+
medianMs: lat.length ? lat[lat.length >> 1] : null,
|
|
118
|
+
errors: answered
|
|
119
|
+
.filter((j) => !satisfies(j.ruling.decision, want(j.r)))
|
|
120
|
+
.map((j) => `said "${j.ruling.decision}" where "${want(j.r)}" was right — ${j.r.expect.slice(0, 48)}`),
|
|
121
|
+
});
|
|
122
|
+
process.stdout.write(' done\n');
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const stillRule = rows.filter((r) => {
|
|
126
|
+
const d = (r.situation.stillMs ?? 0) > STILL_MS_THRESHOLD ? 'stop' : 'wait';
|
|
127
|
+
return satisfies(d, want(r));
|
|
128
|
+
}).length / rows.length;
|
|
129
|
+
|
|
130
|
+
console.log(`\n${'arm'.padEnd(26)} ${'accuracy'.padStart(9)} ${'median'.padStart(9)} ${'no answer'.padStart(10)}`);
|
|
131
|
+
console.log('-'.repeat(58));
|
|
132
|
+
console.log(`${`always "${commonest[0]}"`.padEnd(26)} ${pct(baseline).padStart(9)} ${'—'.padStart(9)} ${'—'.padStart(10)}`);
|
|
133
|
+
console.log(`${`stillMs > ${STILL_MS_THRESHOLD}ms`.padEnd(26)} ${pct(stillRule).padStart(9)} ${'0ms'.padStart(9)} ${'—'.padStart(10)}`);
|
|
134
|
+
for (const r of results) {
|
|
135
|
+
console.log(`${r.arm.padEnd(26)} ${pct(r.accuracy).padStart(9)} ${`${r.medianMs ?? '—'}ms`.padStart(9)} ${`${r.unanswered}`.padStart(10)}`);
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
if (withAbstain) {
|
|
139
|
+
console.log(`\n--- item 100: the fourth word ---`);
|
|
140
|
+
console.log('The question is not whether it abstains. It is whether it abstains on');
|
|
141
|
+
console.log('the ones it would have got WRONG, or at random. A judge that declines');
|
|
142
|
+
console.log('uniformly has added latency and a round trip and bought nothing.\n');
|
|
143
|
+
console.log(`${'arm'.padEnd(22)} ${'answered'.padStart(9)} ${'of those'.padStart(9)} ${'abstained'.padStart(10)} ${'escalations'.padStart(12)}`);
|
|
144
|
+
console.log('-'.repeat(66));
|
|
145
|
+
for (const r of results) {
|
|
146
|
+
const answered = r.judged.filter((j) => j.ruling);
|
|
147
|
+
const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
|
|
148
|
+
const abstained = r.judged.filter((j) => j.abstained).length;
|
|
149
|
+
console.log(`${r.arm.padEnd(22)} ${`${answered.length}/${rows.length}`.padStart(9)} ${pct(right / (answered.length || 1)).padStart(9)} ${`${abstained}`.padStart(10)} ${pct(abstained / rows.length).padStart(12)}`);
|
|
150
|
+
}
|
|
151
|
+
console.log('\n`of those` is accuracy on the questions it chose to answer. If the fourth');
|
|
152
|
+
console.log('word is working, that number is higher than the three-word accuracy above');
|
|
153
|
+
console.log('by more than the abstention rate would give by chance.');
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
console.log('\nerrors, by arm:');
|
|
157
|
+
for (const r of results) {
|
|
158
|
+
console.log(` ${r.arm}`);
|
|
159
|
+
if (!r.errors.length) console.log(' none');
|
|
160
|
+
const seen = new Map();
|
|
161
|
+
for (const e of r.errors) seen.set(e, (seen.get(e) ?? 0) + 1);
|
|
162
|
+
for (const [e, n] of [...seen].sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${e}`);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// --- the cascade: threshold -> local model -> Claude -----------------------
|
|
166
|
+
//
|
|
167
|
+
// The owner's proposal, and the numbers are the only way to say whether it
|
|
168
|
+
// helps: answer with the free rule where it is confident, fall through to the
|
|
169
|
+
// on-device model where it is not, and only then pay a round trip.
|
|
170
|
+
//
|
|
171
|
+
// **A cascade needs a tier that can decline, and neither tier has one.** The
|
|
172
|
+
// threshold is a comparison — it always answers. The supervisor's vocabulary is
|
|
173
|
+
// three words and none of them is "I don't know". So the interesting number is
|
|
174
|
+
// not "does a cascade help" but "how much would an abstain token be worth", and
|
|
175
|
+
// that is item 100. This measures it directly, by letting the rule abstain in a
|
|
176
|
+
// band around its own threshold and handing those to the next tier.
|
|
177
|
+
const armDecisions = new Map(results.map((r) => [r.arm, r.judged]));
|
|
178
|
+
console.log('\n--- cascade: the rule answers, the model covers where it abstains ---');
|
|
179
|
+
console.log(`${'abstain band'.padEnd(22)} ${'rule'.padStart(6)} ${'->model'.padStart(8)} ${'cascade'.padStart(9)} ${'model calls'.padStart(12)}`);
|
|
180
|
+
for (const band of [0, 250, 500, 750, 1000, 1500]) {
|
|
181
|
+
const lo = STILL_MS_THRESHOLD - band;
|
|
182
|
+
const hi = STILL_MS_THRESHOLD + band;
|
|
183
|
+
for (const arm of arms) {
|
|
184
|
+
const judged = armDecisions.get(arm) ?? [];
|
|
185
|
+
let right = 0;
|
|
186
|
+
let escalated = 0;
|
|
187
|
+
for (const [i, r] of rows.entries()) {
|
|
188
|
+
const still = r.situation.stillMs ?? 0;
|
|
189
|
+
if (band > 0 && still >= lo && still <= hi) {
|
|
190
|
+
escalated += 1;
|
|
191
|
+
const ruling = judged[i]?.ruling;
|
|
192
|
+
if (ruling && satisfies(ruling.decision, want(r))) right += 1;
|
|
193
|
+
continue;
|
|
194
|
+
}
|
|
195
|
+
if (satisfies(still > STILL_MS_THRESHOLD ? 'stop' : 'wait', want(r))) right += 1;
|
|
196
|
+
}
|
|
197
|
+
console.log(`${`+/-${band}ms -> ${arm}`.padEnd(22)} ${pct(stillRule).padStart(6)} ${`${escalated}`.padStart(8)} ${pct(right / rows.length).padStart(9)} ${`${escalated}/${rows.length}`.padStart(12)}`);
|
|
198
|
+
}
|
|
199
|
+
if (band === 0) console.log(' (band 0 = no abstention, the rule alone — every row below adds a tier)');
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
console.log('\nThis scores DECISIONS on identical inputs. It cannot score outcomes —');
|
|
203
|
+
console.log('whether acting on a ruling recovered the flow is a fact about the device at');
|
|
204
|
+
console.log('that moment, and belongs to whichever arm was live. See docs/EXPERIMENTS.md.');
|
|
205
|
+
|
|
206
|
+
supervisor.close();
|
|
@@ -42,7 +42,25 @@ const all = metrics.readSupervisions(dev.udid).filter((r) => CORRECT[r.expect]);
|
|
|
42
42
|
// test was 16/16. Mixing a known-bad population into the denominator is the
|
|
43
43
|
// same error as before wearing a different hat.
|
|
44
44
|
const lastN = Number(arg('last', 0));
|
|
45
|
-
|
|
45
|
+
// Which judge. Every arm of the capacity comparison writes to one log, so
|
|
46
|
+
// scoring without this would average a 3B, an 8B and a 14B into a single
|
|
47
|
+
// meaningless number — and it would look like a result.
|
|
48
|
+
const arm = arg('arm', null);
|
|
49
|
+
const armOf = (r) => r.supervisor ?? 'unrecorded';
|
|
50
|
+
const scoped = arm ? all.filter((r) => armOf(r) === arm) : all;
|
|
51
|
+
const rows = lastN > 0 ? scoped.slice(-lastN) : scoped;
|
|
52
|
+
|
|
53
|
+
// A population spanning several arms is not one population. Say so, and say
|
|
54
|
+
// what to pass, rather than printing an average of things that were never
|
|
55
|
+
// comparable.
|
|
56
|
+
const arms = [...new Set(rows.map(armOf))];
|
|
57
|
+
if (arms.length > 1) {
|
|
58
|
+
console.log(`this log holds rulings from ${arms.length} supervisor arms:`);
|
|
59
|
+
for (const a of arms) console.log(` ${String(rows.filter((r) => armOf(r) === a).length).padStart(4)} ${a}`);
|
|
60
|
+
console.log('\nScore one at a time — `--arm=<name>` — because averaging them is not a result.');
|
|
61
|
+
process.exit(2);
|
|
62
|
+
}
|
|
63
|
+
if (rows.length) console.log(`arm: ${arms[0]}\n`);
|
|
46
64
|
if (rows.length < 8) {
|
|
47
65
|
console.error(`only ${rows.length} labelled ruling(s) — run scripts/collect-rulings.mjs first`);
|
|
48
66
|
process.exit(2);
|