simframe 0.12.2 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +138 -9
- package/native/simframed/Sources/SimframeCore/Motion.swift +33 -2
- package/native/supervise.swift +63 -4
- package/package.json +1 -1
- package/scripts/article-md.mjs +185 -0
- package/scripts/ci-device-guard.mjs +82 -0
- package/scripts/ci-integration-local.sh +33 -9
- package/scripts/ci-memory.mjs +117 -15
- package/scripts/eval-fingerprint.mjs +145 -3
- package/scripts/replay-rulings.mjs +60 -1
- package/src/actions.js +213 -13
- package/src/analyze.js +56 -0
- package/src/cli.js +205 -15
- package/src/fingerprint.js +10 -1
- package/src/graph.js +15 -2
- package/src/index.js +302 -16
- package/src/input.js +4 -0
- package/src/matching.js +17 -1
- package/src/mcp.js +148 -13
- package/src/metrics.js +49 -6
- package/src/navigate.js +4 -1
- package/src/ollama.js +37 -9
- package/src/platform/android.js +34 -0
- package/src/platform/index.js +7 -0
- package/src/platform/ios.js +128 -5
- package/src/platform/plist.js +156 -0
- package/src/refs.js +12 -1
- package/src/regions.js +54 -0
- package/src/screenmap.js +85 -6
- package/src/storage.js +201 -0
- package/src/store.js +53 -0
- package/src/supervisor.js +20 -2
- package/src/view.js +84 -9
|
@@ -43,6 +43,15 @@ for (const key of ["input.driver", "ax.driver"]) {
|
|
|
43
43
|
console.log(`${good ? "ok " : "FAIL"} ${key} = ${JSON.stringify(d[key])} (want "simframed")`);
|
|
44
44
|
if (!good) failed = true;
|
|
45
45
|
}
|
|
46
|
+
// A configured driver is not an answering driver — the same check as ci.yml.
|
|
47
|
+
// A whole CI run read the screen eighteen times, every reading came back
|
|
48
|
+
// OCR-only, and this step said `ok` because a driver was present.
|
|
49
|
+
{
|
|
50
|
+
const n = d["ax.elements"];
|
|
51
|
+
const good = typeof n === "number" && n > 0;
|
|
52
|
+
console.log(`${good ? "ok " : "FAIL"} ax.elements = ${JSON.stringify(n)} (want > 0 — the tree must answer, not merely exist)`);
|
|
53
|
+
if (!good) failed = true;
|
|
54
|
+
}
|
|
46
55
|
if (d.warnings > 0) {
|
|
47
56
|
console.log(`\n${d.warnings} degraded layer(s):`);
|
|
48
57
|
for (const c of d.checks.filter((c) => c.level !== "ok")) console.log(` ${c.level} ${c.name}: ${c.detail}`);
|
|
@@ -58,17 +67,32 @@ read_state() {
|
|
|
58
67
|
printf '[{"button":"home"},{"settle":true}]\n' > /tmp/reset-local.json
|
|
59
68
|
node src/cli.js do /tmp/reset-local.json --device="$DEVICE" >/dev/null 2>&1
|
|
60
69
|
BEFORE=$(read_state)
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
70
|
+
# From inside an app, pressing home always changes the screen — see the job's
|
|
71
|
+
# comment for the four vehicles that did not hold. Setup gets to a known screen;
|
|
72
|
+
# the asserted action is simframe's own HID path with no simctl in it.
|
|
73
|
+
printf '[{"launch":{"value":"com.apple.Preferences","relaunch":true}},{"settle":true}]\n' > /tmp/setup-local.json
|
|
74
|
+
node src/cli.js do /tmp/setup-local.json --device="$DEVICE" >/tmp/step-local.log 2>&1 || true
|
|
75
|
+
BEFORE=$(read_state)
|
|
76
|
+
printf '[{"button":"home"},{"settle":true}]\n' > /tmp/flow-local.json
|
|
77
|
+
node src/cli.js do /tmp/flow-local.json --device="$DEVICE" >>/tmp/step-local.log 2>&1 || true
|
|
78
|
+
AFTER=$(read_state)
|
|
79
|
+
if [ "${BEFORE%% *}" != "${AFTER%% *}" ]; then ok "frame hash changed: ${BEFORE%% *} -> ${AFTER%% *}"
|
|
80
|
+
else bad "a step ran and capture saw no change"; tail -5 /tmp/step-local.log; fi
|
|
69
81
|
|
|
70
82
|
step "The memory layer — screen map, refs, graph, verdicts, flows"
|
|
71
|
-
|
|
83
|
+
# Mirrors the job: exit 75 means the display wedged and nothing was tested, so
|
|
84
|
+
# revive once and run again; any other non-zero is a real check failing. The
|
|
85
|
+
# wedge lands on an app switch and hit roughly every other run of this script on
|
|
86
|
+
# the day it was written, which is most of why the job looked flaky.
|
|
87
|
+
if node scripts/ci-memory.mjs --device="$DEVICE"; then
|
|
88
|
+
ok "ci-memory"
|
|
89
|
+
elif [ $? = 75 ]; then
|
|
90
|
+
printf ' (the display wedged — DEFERRED 126. Reviving once and running again.)\n'
|
|
91
|
+
node src/cli.js revive --device="$DEVICE" || true
|
|
92
|
+
node scripts/ci-memory.mjs --device="$DEVICE" && ok "ci-memory (after one revive)" || bad "ci-memory"
|
|
93
|
+
else
|
|
94
|
+
bad "ci-memory"
|
|
95
|
+
fi
|
|
72
96
|
|
|
73
97
|
step "The fingerprint distributions, measured and bounded"
|
|
74
98
|
node scripts/eval-fingerprint.mjs --tour=test/tours/device-native.json --rounds=3 \
|
package/scripts/ci-memory.mjs
CHANGED
|
@@ -50,6 +50,9 @@ let failures = 0;
|
|
|
50
50
|
*/
|
|
51
51
|
let deviceDied = null;
|
|
52
52
|
|
|
53
|
+
/** EX_TEMPFAIL: the device died under the checks, so nothing was tested. */
|
|
54
|
+
const DEVICE_DIED_EXIT = 75;
|
|
55
|
+
|
|
53
56
|
function check(ok, label, detail = '') {
|
|
54
57
|
if (!ok) failures += 1;
|
|
55
58
|
console.log(`${ok ? 'ok ' : 'FAIL'} ${label}${detail ? ` — ${detail}` : ''}`);
|
|
@@ -108,7 +111,14 @@ async function cli(args, { expectFail = false, allowFail = false } = {}) {
|
|
|
108
111
|
// detail has been truncated for legibility, and the first version of this
|
|
109
112
|
// guard looked for "did not produce a frame" in a string that had been cut
|
|
110
113
|
// to "simframe daemon di". The full text only exists at this boundary.
|
|
111
|
-
|
|
114
|
+
//
|
|
115
|
+
// And when the payload is a JSON report, say what failed rather than
|
|
116
|
+
// handing back its first hundred characters. A run of this printed
|
|
117
|
+
// `simframe doctor --json failed: {\n "ok": false,\n "strict": true,\n
|
|
118
|
+
// "failu` — the word "failures" cut in half, one character before the only
|
|
119
|
+
// content that mattered. A harness that truncates away the reason is doing
|
|
120
|
+
// to its reader exactly what this repo keeps writing items about.
|
|
121
|
+
throw new Error(`simframe ${full.join(' ')} failed: ${summarise(why)}`);
|
|
112
122
|
}
|
|
113
123
|
}
|
|
114
124
|
|
|
@@ -171,13 +181,42 @@ async function jsonRetry(args, opts, attempts = 3) {
|
|
|
171
181
|
console.error(` ${String(last.message).split('\n')[0]}`);
|
|
172
182
|
console.error('\nEverything after this point would be testing a dead simulator, so the run');
|
|
173
183
|
console.error('stops here. This is not a memory-layer failure — it is the device-state');
|
|
174
|
-
console.error('problem in docs/DEFERRED.md. A device restart is the only known cure
|
|
175
|
-
console.error('
|
|
176
|
-
|
|
184
|
+
console.error('problem in docs/DEFERRED.md (126). A device restart is the only known cure,');
|
|
185
|
+
console.error('and `simframe revive` is that restart.');
|
|
186
|
+
// Exit 75, not 1, and the distinction is the whole point of this file.
|
|
187
|
+
//
|
|
188
|
+
// "A check about the memory layer failed" and "the simulator died under the
|
|
189
|
+
// checks" are different conditions with different responses, and for two CI
|
|
190
|
+
// rounds they were one exit code — so a caller could only retry everything
|
|
191
|
+
// or retry nothing. 75 is EX_TEMPFAIL, which is exactly what this is: the
|
|
192
|
+
// subject under test was never reached.
|
|
193
|
+
//
|
|
194
|
+
// The caller reviving and running again is not papering over a product bug.
|
|
195
|
+
// The wedge is a documented CoreSimulator condition, `frame --fresh` names
|
|
196
|
+
// it, and `revive` is the cure this project ships for it — CI simply had no
|
|
197
|
+
// way to say "use it".
|
|
198
|
+
process.exit(DEVICE_DIED_EXIT);
|
|
177
199
|
}
|
|
178
200
|
throw last;
|
|
179
201
|
}
|
|
180
202
|
|
|
203
|
+
/** A failed JSON report, reduced to the part that says what went wrong. */
|
|
204
|
+
function summarise(why) {
|
|
205
|
+
try {
|
|
206
|
+
const parsed = JSON.parse(why);
|
|
207
|
+
const failures = parsed.failures ?? parsed.failing ?? null;
|
|
208
|
+
if (Array.isArray(failures) && failures.length) {
|
|
209
|
+
return failures
|
|
210
|
+
.map((f) => (typeof f === 'string' ? f : `${f.name ?? f.check ?? '?'}: ${f.detail ?? f.note ?? f.message ?? ''}`.trim()))
|
|
211
|
+
.join('; ')
|
|
212
|
+
.slice(0, 400);
|
|
213
|
+
}
|
|
214
|
+
const bad = (parsed.checks ?? []).filter((c) => c.ok === false);
|
|
215
|
+
if (bad.length) return bad.map((c) => `${c.name}: ${c.detail ?? ''}`.trim()).join('; ').slice(0, 400);
|
|
216
|
+
} catch { /* not JSON, or not a shape we know — fall through to the raw text */ }
|
|
217
|
+
return why.slice(0, 400);
|
|
218
|
+
}
|
|
219
|
+
|
|
181
220
|
const markHash = async () => (await jsonRetry(['mark'])).hash;
|
|
182
221
|
|
|
183
222
|
function writeFlow(name, steps) {
|
|
@@ -244,7 +283,36 @@ if (failures) {
|
|
|
244
283
|
|
|
245
284
|
console.log('\n--- the screen map ---');
|
|
246
285
|
await jsonRetry(['do', LOOP], { allowFail: true });
|
|
247
|
-
const map = await
|
|
286
|
+
const map = await readableMap();
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* A map that is empty *and* says a sensor failed is a blink, not a result.
|
|
290
|
+
*
|
|
291
|
+
* `jsonRetry` retries a command that throws, and this one did not throw: `ui`
|
|
292
|
+
* returned 200 with zero elements and a degraded note, which is the shape of
|
|
293
|
+
* failure this whole repo keeps writing rules about. The run that made this
|
|
294
|
+
* necessary had the accessibility read time out once, report an empty map, and
|
|
295
|
+
* then resolve a ref correctly sixty seconds later on the same device — so the
|
|
296
|
+
* device was fine and the check had caught one bad read.
|
|
297
|
+
*
|
|
298
|
+
* Deliberately narrow. An empty map with **no** degraded sensor is a real
|
|
299
|
+
* answer — that is a blank screen and the check should fail on it. Only an
|
|
300
|
+
* empty map that admits a layer did not answer is worth asking again, and after
|
|
301
|
+
* three attempts it fails with what the sensor said, which is the diagnosis
|
|
302
|
+
* either way.
|
|
303
|
+
*/
|
|
304
|
+
async function readableMap(attempts = 3) {
|
|
305
|
+
let last;
|
|
306
|
+
for (let i = 0; i < attempts; i += 1) {
|
|
307
|
+
last = await jsonRetry(['ui']);
|
|
308
|
+
if (last.elements?.length || !(last.degraded ?? []).length) return last;
|
|
309
|
+
if (i < attempts - 1) {
|
|
310
|
+
console.log(` (the map came back empty and a sensor said why — retrying \`ui\`: ${(last.degraded ?? []).join('; ')})`);
|
|
311
|
+
await new Promise((r) => setTimeout(r, 2000));
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
return last;
|
|
315
|
+
}
|
|
248
316
|
|
|
249
317
|
// Report what actually answered rather than asserting the runner's situation.
|
|
250
318
|
// This line used to read "with no accessibility tree available" unconditionally,
|
|
@@ -263,14 +331,38 @@ check(Number.isFinite(map.points?.width) && Number.isFinite(map.points?.height),
|
|
|
263
331
|
'the map knows the screen size in points', `${map.points?.width}x${map.points?.height}pt`);
|
|
264
332
|
|
|
265
333
|
const refs = (map.elements ?? []).map((e) => e.ref);
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
334
|
+
// Three checks over a collection, and `every`/`!some` are all true of an empty
|
|
335
|
+
// one. On a CI run whose map came back with **0 elements** they printed
|
|
336
|
+
// `ok refs are numbered 1..n with no gaps — #1..#0` and two more like it:
|
|
337
|
+
// three lines of reassurance about nothing, directly under the failure that
|
|
338
|
+
// said the map was empty.
|
|
339
|
+
//
|
|
340
|
+
// That is the exact defect three field reports spent a day describing — a
|
|
341
|
+
// confident statement that verified nothing — and the harness was doing it to
|
|
342
|
+
// itself in the same output. Untested is not passed.
|
|
343
|
+
if (!map.elements?.length) {
|
|
344
|
+
for (const label of [
|
|
345
|
+
'refs are numbered 1..n with no gaps',
|
|
346
|
+
'every element has a tap point on the screen',
|
|
347
|
+
'the status bar is not offered as something to tap',
|
|
348
|
+
]) skip(label, 'the map was empty, so there was nothing to check');
|
|
349
|
+
// And say what the device was showing, because an empty map on a live device
|
|
350
|
+
// is the shape this project has chased under four different symptoms. The
|
|
351
|
+
// liveness note carries the ignored-gesture check; `stableForMs` and the
|
|
352
|
+
// frame age are the two numbers whose disagreement names a dead surface.
|
|
353
|
+
const st = await jsonRetry(['state'], { allowFail: true });
|
|
354
|
+
console.log(` the device at that moment: frame #${st?.seq ?? '?'}, `
|
|
355
|
+
+ `${st?.stableForMs ?? '?'}ms still, ${st?.live?.note ?? 'liveness reported nothing'}`);
|
|
356
|
+
} else {
|
|
357
|
+
check(refs.every((r, i) => r === i + 1),
|
|
358
|
+
'refs are numbered 1..n with no gaps', `#1..#${refs.length}`);
|
|
359
|
+
check(map.elements.every((e) =>
|
|
360
|
+
Number.isInteger(e.x) && Number.isInteger(e.y)
|
|
361
|
+
&& e.y >= 0 && e.y <= map.points.height && e.x >= 0 && e.x <= map.points.width),
|
|
362
|
+
'every element has a tap point on the screen');
|
|
363
|
+
check(!map.elements.some((e) => e.region === 'status-bar'),
|
|
364
|
+
'the status bar is not offered as something to tap');
|
|
365
|
+
}
|
|
274
366
|
|
|
275
367
|
console.log('\n--- element refs ---');
|
|
276
368
|
// Re-read on a screen we chose, rather than on whatever the device happened to
|
|
@@ -296,7 +388,8 @@ if (first) {
|
|
|
296
388
|
// hashes only have to agree with that, and they only get a say when they are
|
|
297
389
|
// informative enough to have one.
|
|
298
390
|
const moved = !informativeHash(before) || !informativeHash(after) || before !== after;
|
|
299
|
-
|
|
391
|
+
const launched = ran(left);
|
|
392
|
+
if (!launched) skip('the screen actually changed before testing the stale ref',
|
|
300
393
|
'the second app never launched, so there was no screen change to test against');
|
|
301
394
|
else check(moved, 'the screen actually changed before testing the stale ref',
|
|
302
395
|
`${before.slice(0, 10)} -> ${after.slice(0, 10)}`
|
|
@@ -305,7 +398,16 @@ if (first) {
|
|
|
305
398
|
// reports "the stale-ref guard failed" for a device that never left the
|
|
306
399
|
// screen, which is a false accusation against the one layer this file exists
|
|
307
400
|
// to defend — and it is how this check has failed twice.
|
|
308
|
-
|
|
401
|
+
// A skip has to propagate. The precondition above reported NOT TESTED and this
|
|
402
|
+
// check ran anyway and failed — which is the harness doing to itself, one line
|
|
403
|
+
// later, exactly what `skip` was written to stop it doing. `moved` is true
|
|
404
|
+
// when a hash is too degenerate to have a say, and that is right for "did the
|
|
405
|
+
// screen change" and wrong as a licence to run a check whose setup is known
|
|
406
|
+
// not to have happened.
|
|
407
|
+
if (!launched) {
|
|
408
|
+
skip('a ref numbered on another screen refuses instead of tapping those coordinates',
|
|
409
|
+
'we never reached another screen, so there was nothing to refuse from');
|
|
410
|
+
} else if (moved) {
|
|
309
411
|
// This matched on prose twice and went red twice, both times for a refusal
|
|
310
412
|
// that was correct and better worded than the alternation knew — most
|
|
311
413
|
// recently `"Welcome to Reminders" is not on this screen`, which refuses
|
|
@@ -86,6 +86,17 @@ console.log(`tour: ${tour.length} screens x ${rounds} rounds, every reading cold
|
|
|
86
86
|
const readings = [];
|
|
87
87
|
/** Navigations that did not land before the reading was taken. */
|
|
88
88
|
const arrivalFailures = [];
|
|
89
|
+
/** Readings too bare to be a screen — see the guard where this is used. */
|
|
90
|
+
const sparseReadings = [];
|
|
91
|
+
/**
|
|
92
|
+
* Below this, a reading cannot distinguish its screen from any other bare one.
|
|
93
|
+
*
|
|
94
|
+
* Not a failure threshold — see where it is used. The Settings root legitimately
|
|
95
|
+
* reads 4 tokens on a hosted runner, so treating this as "the screen has not
|
|
96
|
+
* drawn" rejected a real screen and made CI deterministically red. It marks
|
|
97
|
+
* readings worth suspecting when a score disagrees with itself, nothing more.
|
|
98
|
+
*/
|
|
99
|
+
const MIN_TOKENS_FOR_A_READING = 5;
|
|
89
100
|
|
|
90
101
|
/**
|
|
91
102
|
* The tokens that carry a name, as opposed to a shape.
|
|
@@ -124,6 +135,22 @@ const save = (extra = {}) => {
|
|
|
124
135
|
}, null, 2));
|
|
125
136
|
};
|
|
126
137
|
|
|
138
|
+
/**
|
|
139
|
+
* How hard to try for a frame newer than the navigation before giving up.
|
|
140
|
+
*
|
|
141
|
+
* Three re-reads a second apart is ~3s of slack against capture medians that
|
|
142
|
+
* were measured above 1s on a bad runner. Generous enough to absorb the spikes
|
|
143
|
+
* that caused this, small enough that a genuinely stopped capture still fails
|
|
144
|
+
* rather than hanging the job.
|
|
145
|
+
*/
|
|
146
|
+
const STALE_READ_RETRIES = 3;
|
|
147
|
+
const STALE_READ_WAIT_MS = 1000;
|
|
148
|
+
|
|
149
|
+
/** When the steps that were supposed to change the screen finished. */
|
|
150
|
+
let navigatedAt = 0;
|
|
151
|
+
/** Readings that never got a frame newer than their own navigation. */
|
|
152
|
+
const staleReadings = [];
|
|
153
|
+
|
|
127
154
|
for (let round = 1; round <= rounds; round += 1) {
|
|
128
155
|
for (const screen of tour) {
|
|
129
156
|
if (screen.steps?.length) {
|
|
@@ -140,6 +167,7 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
140
167
|
// matters.
|
|
141
168
|
try {
|
|
142
169
|
await actions.runScript(device, { steps: screen.steps, verify: false });
|
|
170
|
+
navigatedAt = Date.now();
|
|
143
171
|
} catch (err) {
|
|
144
172
|
save({ abandonedAt: { screen: screen.name, round, error: err.message } });
|
|
145
173
|
console.error(`\nFAIL round ${round}, "${screen.name}" never arrived: ${err.message}`);
|
|
@@ -148,7 +176,66 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
148
176
|
process.exit(1);
|
|
149
177
|
}
|
|
150
178
|
}
|
|
151
|
-
|
|
179
|
+
let id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
180
|
+
// A reading off a frame older than the navigation is not a reading either.
|
|
181
|
+
//
|
|
182
|
+
// Same rule as the sparseness guard below, on the other axis, and it took a
|
|
183
|
+
// third symptom to see they were one cause. Three CI runs failed this step
|
|
184
|
+
// three different ways — `settings` reading 4 tokens, a reading that "does
|
|
185
|
+
// not resemble its own screen", and a reading "taken on the previous
|
|
186
|
+
// screen" at similarity 1.00 off a frame **7168ms old**. All three are the
|
|
187
|
+
// same sentence: the reading is not of the screen we think it is, because
|
|
188
|
+
// capture on a hosted runner is slow. The daemon log for that run shows
|
|
189
|
+
// capture medians of 417ms, 503ms, 996ms, 1106ms and 1132ms against the
|
|
190
|
+
// 55.4ms p50 measured for a healthy runner — 20x, and with damage-driven
|
|
191
|
+
// capture a `fresh` read returns the newest frame that EXISTS, which on a
|
|
192
|
+
// runner that far behind can predate the navigation entirely.
|
|
193
|
+
//
|
|
194
|
+
// So the frame must have been captured after the steps that were supposed
|
|
195
|
+
// to change the screen. Re-read rather than fail, and fail only if it stays
|
|
196
|
+
// stale — an eval that scores a stale frame is measuring the runner, which
|
|
197
|
+
// is the one thing this harness says it is not doing.
|
|
198
|
+
if (navigatedAt) {
|
|
199
|
+
for (let attempt = 0; attempt < STALE_READ_RETRIES; attempt += 1) {
|
|
200
|
+
const capturedAt = id.state?.capturedAt ?? 0;
|
|
201
|
+
if (capturedAt >= navigatedAt) break;
|
|
202
|
+
const age = Date.now() - capturedAt;
|
|
203
|
+
console.log(` ("${screen.name}" read a frame from ${age}ms ago, older than the navigation`
|
|
204
|
+
+ ` — reading again ${attempt + 1}/${STALE_READ_RETRIES})`);
|
|
205
|
+
await new Promise((r) => setTimeout(r, STALE_READ_WAIT_MS));
|
|
206
|
+
id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
207
|
+
}
|
|
208
|
+
if ((id.state?.capturedAt ?? 0) < navigatedAt) {
|
|
209
|
+
staleReadings.push(`${screen.name} round ${round}: frame still predates the navigation`
|
|
210
|
+
+ ` by ${navigatedAt - (id.state?.capturedAt ?? 0)}ms after ${STALE_READ_RETRIES} re-reads`);
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
// A reading too sparse to be a screen is not a reading.
|
|
214
|
+
//
|
|
215
|
+
// On a runner measured at ~2.5x slower than a laptop (EXPERIMENTS §15),
|
|
216
|
+
// perception can land on a half-rendered screen: two CI runs recorded
|
|
217
|
+
// `settings` and `settings-general` **both at 4 tokens**, identical hash,
|
|
218
|
+
// and the eval duly reported that a reading did not resemble its own
|
|
219
|
+
// screen. It resembled nothing, because almost nothing had been drawn yet.
|
|
220
|
+
//
|
|
221
|
+
// This is the hazard `TOKEN_RULES_VERSION` 7 was written for, arriving
|
|
222
|
+
// through the harness instead of the rules: *"two sparse nameless readings
|
|
223
|
+
// then matched exactly, one hash standing for two different screens"*. A
|
|
224
|
+
// guard exists for identity and there was none here.
|
|
225
|
+
//
|
|
226
|
+
// Read again rather than fail, and fail only if it stays sparse — the same
|
|
227
|
+
// "untested is not passed" rule the memory harness learned. A screen that
|
|
228
|
+
// is genuinely this bare after a second look is a real finding.
|
|
229
|
+
if ((id.tokens ?? []).length < MIN_TOKENS_FOR_A_READING) {
|
|
230
|
+
const before = (id.tokens ?? []).length;
|
|
231
|
+
await new Promise((r) => setTimeout(r, 1500));
|
|
232
|
+
id = await api.screenIdentity(device, { fresh: true, confirmNovel: false });
|
|
233
|
+
const after = (id.tokens ?? []).length;
|
|
234
|
+
console.log(` ("${screen.name}" read ${before} token(s) — too sparse to compare; read again: ${after})`);
|
|
235
|
+
if (after < MIN_TOKENS_FOR_A_READING) {
|
|
236
|
+
sparseReadings.push(`${screen.name} round ${round}: ${after} token(s) after two reads`);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
152
239
|
// Did we actually arrive? Two differently-named screens reading the same
|
|
153
240
|
// fingerprint means the navigation did not land before the reading was
|
|
154
241
|
// taken, and every distribution below it is then measuring the tour rather
|
|
@@ -164,8 +251,21 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
164
251
|
// means the navigation did not happen.
|
|
165
252
|
const s = fingerprint.similarity(previous.tokens, id.tokens ?? []);
|
|
166
253
|
if (s >= ARRIVAL_SUSPICION) {
|
|
254
|
+
// What was actually on screen, not just that it was the wrong thing.
|
|
255
|
+
//
|
|
256
|
+
// This check has fired twice on CI and both times the report was a
|
|
257
|
+
// similarity score and two screen names, which is enough to know the
|
|
258
|
+
// run is void and not enough to know why. The tour asserts arrival with
|
|
259
|
+
// a `waitFor` before this ever runs, so a failure here means the
|
|
260
|
+
// `waitFor` *passed* on a screen that was not the destination — and the
|
|
261
|
+
// labels are the only thing that can say what that screen was.
|
|
262
|
+
const labels = (id.entry?.targets ?? [])
|
|
263
|
+
.map((t) => t.label).filter(Boolean).slice(0, 12).map((l) => l.slice(0, 24));
|
|
167
264
|
arrivalFailures.push(
|
|
168
|
-
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
|
|
265
|
+
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
|
|
266
|
+
+ `\n sensors: ${(id.entry?.sources ?? []).join('+') || 'none'}`
|
|
267
|
+
+ `, settled: ${id.settled}, frame ${Math.round(Date.now() - (id.state?.capturedAt ?? Date.now()))}ms old`
|
|
268
|
+
+ `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`);
|
|
169
269
|
}
|
|
170
270
|
}
|
|
171
271
|
readings.push({
|
|
@@ -206,9 +306,45 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
206
306
|
}
|
|
207
307
|
}
|
|
208
308
|
|
|
209
|
-
save();
|
|
309
|
+
save({ staleReadings });
|
|
210
310
|
if (outFile) console.log(`\nwrote ${readings.length} readings to ${outFile}`);
|
|
211
311
|
|
|
312
|
+
if (sparseReadings.length) {
|
|
313
|
+
// Reported, NOT failed — and that correction is worth more than the check.
|
|
314
|
+
//
|
|
315
|
+
// This exited 1 for four hours on 2026-09-14 and turned an intermittent CI
|
|
316
|
+
// failure into a deterministic one. The reasoning was that a 4-token reading
|
|
317
|
+
// is a screen that has not drawn. It is not: the Settings root legitimately
|
|
318
|
+
// reads 4 tokens on a hosted runner, twice in a row, three times in a row —
|
|
319
|
+
// which the comment on `MIN_TOKENS_FOR_A_READING` had itself said ("4-8
|
|
320
|
+
// tokens") one screen above the code that rejected it.
|
|
321
|
+
//
|
|
322
|
+
// What remains true is that a reading this bare cannot distinguish its screen
|
|
323
|
+
// from anything else equally bare, so it is worth saying out loud. What is not
|
|
324
|
+
// true is that saying so should stop the run. The arrival check below is the
|
|
325
|
+
// one that catches the real collision, and it does so on evidence rather than
|
|
326
|
+
// on a token count.
|
|
327
|
+
console.log(`\nNOTE ${sparseReadings.length} reading(s) were too bare to distinguish a screen:`);
|
|
328
|
+
for (const f of sparseReadings) console.log(` ${f}`);
|
|
329
|
+
console.log(`\nFewer than ${MIN_TOKENS_FOR_A_READING} tokens after two reads. That is not automatically`);
|
|
330
|
+
console.log('wrong — a plain screen really can be this bare — but two readings this sparse');
|
|
331
|
+
console.log('cannot be told apart, which is the hazard TOKEN_RULES_VERSION 7 was written');
|
|
332
|
+
console.log('for. If a same-screen score below disagrees with itself, start here.');
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
if (staleReadings.length) {
|
|
336
|
+
// A NOTE and not a failure, for the same reason the sparseness guard is one:
|
|
337
|
+
// this says the runner was too slow to give a fresh frame, which is a fact
|
|
338
|
+
// about the machine and not about the fingerprint. It is printed *above* the
|
|
339
|
+
// arrival check on purpose — when both fire, this is the explanation of that.
|
|
340
|
+
console.log(`\nNOTE ${staleReadings.length} reading(s) never got a frame newer than their navigation:`);
|
|
341
|
+
for (const f of staleReadings) console.log(` ${f}`);
|
|
342
|
+
console.log(`\nCapture on this host is behind the tour. With damage-driven capture a "fresh"`);
|
|
343
|
+
console.log('read returns the newest frame that exists, so on a slow runner it can predate');
|
|
344
|
+
console.log('the navigation entirely. If an arrival failure follows, this is its cause and');
|
|
345
|
+
console.log('the tour is not what needs fixing.');
|
|
346
|
+
}
|
|
347
|
+
|
|
212
348
|
if (arrivalFailures.length) {
|
|
213
349
|
console.error(`\nFAIL ${arrivalFailures.length} reading(s) were taken on the previous screen:`);
|
|
214
350
|
for (const f of arrivalFailures) console.error(` ${f}`);
|
|
@@ -216,6 +352,12 @@ if (arrivalFailures.length) {
|
|
|
216
352
|
console.error('measure the tour rather than the fingerprint. Fix the tour and re-run.');
|
|
217
353
|
console.error('A `settle` step defaults to mode "stable", which returns instantly in the');
|
|
218
354
|
console.error('moment before an animation begins — action steps already settle on their own.');
|
|
355
|
+
console.error('');
|
|
356
|
+
console.error('The tour asserts arrival with a `waitFor` before any reading is taken, so a');
|
|
357
|
+
console.error('failure here means that wait PASSED on a screen that was not the destination.');
|
|
358
|
+
console.error('Read the labels above before changing the tour: either the wait matched');
|
|
359
|
+
console.error('something it should not have, or the reading came off a frame older than the');
|
|
360
|
+
console.error('navigation — and those have opposite fixes.');
|
|
219
361
|
process.exit(1);
|
|
220
362
|
}
|
|
221
363
|
|
|
@@ -76,6 +76,8 @@ if (baseline > 0.65) {
|
|
|
76
76
|
const STILL_MS_THRESHOLD = 3000;
|
|
77
77
|
console.log(`\nthe brief every model arm gets is ${ollama.readBrief().length} characters, read from native/supervise.swift`);
|
|
78
78
|
|
|
79
|
+
const withAbstain = process.argv.includes('--abstain');
|
|
80
|
+
|
|
79
81
|
const results = [];
|
|
80
82
|
for (const arm of arms) {
|
|
81
83
|
const judged = [];
|
|
@@ -87,6 +89,7 @@ for (const arm of arms) {
|
|
|
87
89
|
const detail = {};
|
|
88
90
|
const ruling = await supervisor.judge({
|
|
89
91
|
...r.situation,
|
|
92
|
+
mayAbstain: withAbstain,
|
|
90
93
|
options: { supervisor: arm },
|
|
91
94
|
// Wide on purpose. The shipped budget is 2.5s and a 14B will exceed it;
|
|
92
95
|
// capping here would score the larger model on *latency* while calling it
|
|
@@ -96,7 +99,7 @@ for (const arm of arms) {
|
|
|
96
99
|
detail,
|
|
97
100
|
});
|
|
98
101
|
if (!ruling) unanswered += 1;
|
|
99
|
-
judged.push({ r, ruling, detail });
|
|
102
|
+
judged.push({ r, ruling, detail, abstained: detail.kind === 'abstained' });
|
|
100
103
|
process.stdout.write('.');
|
|
101
104
|
}
|
|
102
105
|
const answered = judged.filter((j) => j.ruling);
|
|
@@ -104,6 +107,7 @@ for (const arm of arms) {
|
|
|
104
107
|
const lat = answered.map((j) => j.ruling.ms).filter(Number.isFinite).sort((a, b) => a - b);
|
|
105
108
|
results.push({
|
|
106
109
|
arm,
|
|
110
|
+
judged,
|
|
107
111
|
n: rows.length,
|
|
108
112
|
unanswered,
|
|
109
113
|
// Scored over every situation, not only the answered ones. A judge that
|
|
@@ -131,6 +135,24 @@ for (const r of results) {
|
|
|
131
135
|
console.log(`${r.arm.padEnd(26)} ${pct(r.accuracy).padStart(9)} ${`${r.medianMs ?? '—'}ms`.padStart(9)} ${`${r.unanswered}`.padStart(10)}`);
|
|
132
136
|
}
|
|
133
137
|
|
|
138
|
+
if (withAbstain) {
|
|
139
|
+
console.log(`\n--- item 100: the fourth word ---`);
|
|
140
|
+
console.log('The question is not whether it abstains. It is whether it abstains on');
|
|
141
|
+
console.log('the ones it would have got WRONG, or at random. A judge that declines');
|
|
142
|
+
console.log('uniformly has added latency and a round trip and bought nothing.\n');
|
|
143
|
+
console.log(`${'arm'.padEnd(22)} ${'answered'.padStart(9)} ${'of those'.padStart(9)} ${'abstained'.padStart(10)} ${'escalations'.padStart(12)}`);
|
|
144
|
+
console.log('-'.repeat(66));
|
|
145
|
+
for (const r of results) {
|
|
146
|
+
const answered = r.judged.filter((j) => j.ruling);
|
|
147
|
+
const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
|
|
148
|
+
const abstained = r.judged.filter((j) => j.abstained).length;
|
|
149
|
+
console.log(`${r.arm.padEnd(22)} ${`${answered.length}/${rows.length}`.padStart(9)} ${pct(right / (answered.length || 1)).padStart(9)} ${`${abstained}`.padStart(10)} ${pct(abstained / rows.length).padStart(12)}`);
|
|
150
|
+
}
|
|
151
|
+
console.log('\n`of those` is accuracy on the questions it chose to answer. If the fourth');
|
|
152
|
+
console.log('word is working, that number is higher than the three-word accuracy above');
|
|
153
|
+
console.log('by more than the abstention rate would give by chance.');
|
|
154
|
+
}
|
|
155
|
+
|
|
134
156
|
console.log('\nerrors, by arm:');
|
|
135
157
|
for (const r of results) {
|
|
136
158
|
console.log(` ${r.arm}`);
|
|
@@ -140,6 +162,43 @@ for (const r of results) {
|
|
|
140
162
|
for (const [e, n] of [...seen].sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${e}`);
|
|
141
163
|
}
|
|
142
164
|
|
|
165
|
+
// --- the cascade: threshold -> local model -> Claude -----------------------
|
|
166
|
+
//
|
|
167
|
+
// The owner's proposal, and the numbers are the only way to say whether it
|
|
168
|
+
// helps: answer with the free rule where it is confident, fall through to the
|
|
169
|
+
// on-device model where it is not, and only then pay a round trip.
|
|
170
|
+
//
|
|
171
|
+
// **A cascade needs a tier that can decline, and neither tier has one.** The
|
|
172
|
+
// threshold is a comparison — it always answers. The supervisor's vocabulary is
|
|
173
|
+
// three words and none of them is "I don't know". So the interesting number is
|
|
174
|
+
// not "does a cascade help" but "how much would an abstain token be worth", and
|
|
175
|
+
// that is item 100. This measures it directly, by letting the rule abstain in a
|
|
176
|
+
// band around its own threshold and handing those to the next tier.
|
|
177
|
+
const armDecisions = new Map(results.map((r) => [r.arm, r.judged]));
|
|
178
|
+
console.log('\n--- cascade: the rule answers, the model covers where it abstains ---');
|
|
179
|
+
console.log(`${'abstain band'.padEnd(22)} ${'rule'.padStart(6)} ${'->model'.padStart(8)} ${'cascade'.padStart(9)} ${'model calls'.padStart(12)}`);
|
|
180
|
+
for (const band of [0, 250, 500, 750, 1000, 1500]) {
|
|
181
|
+
const lo = STILL_MS_THRESHOLD - band;
|
|
182
|
+
const hi = STILL_MS_THRESHOLD + band;
|
|
183
|
+
for (const arm of arms) {
|
|
184
|
+
const judged = armDecisions.get(arm) ?? [];
|
|
185
|
+
let right = 0;
|
|
186
|
+
let escalated = 0;
|
|
187
|
+
for (const [i, r] of rows.entries()) {
|
|
188
|
+
const still = r.situation.stillMs ?? 0;
|
|
189
|
+
if (band > 0 && still >= lo && still <= hi) {
|
|
190
|
+
escalated += 1;
|
|
191
|
+
const ruling = judged[i]?.ruling;
|
|
192
|
+
if (ruling && satisfies(ruling.decision, want(r))) right += 1;
|
|
193
|
+
continue;
|
|
194
|
+
}
|
|
195
|
+
if (satisfies(still > STILL_MS_THRESHOLD ? 'stop' : 'wait', want(r))) right += 1;
|
|
196
|
+
}
|
|
197
|
+
console.log(`${`+/-${band}ms -> ${arm}`.padEnd(22)} ${pct(stillRule).padStart(6)} ${`${escalated}`.padStart(8)} ${pct(right / rows.length).padStart(9)} ${`${escalated}/${rows.length}`.padStart(12)}`);
|
|
198
|
+
}
|
|
199
|
+
if (band === 0) console.log(' (band 0 = no abstention, the rule alone — every row below adds a tier)');
|
|
200
|
+
}
|
|
201
|
+
|
|
143
202
|
console.log('\nThis scores DECISIONS on identical inputs. It cannot score outcomes —');
|
|
144
203
|
console.log('whether acting on a ruling recovered the flow is a fact about the device at');
|
|
145
204
|
console.log('that moment, and belongs to whichever arm was live. See docs/EXPERIMENTS.md.');
|