simframe 0.12.2 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +123 -9
- package/native/supervise.swift +63 -4
- package/package.json +1 -1
- package/scripts/article-md.mjs +185 -0
- package/scripts/ci-integration-local.sh +22 -1
- package/scripts/ci-memory.mjs +92 -14
- package/scripts/eval-fingerprint.mjs +20 -1
- package/scripts/replay-rulings.mjs +60 -1
- package/src/actions.js +96 -9
- package/src/analyze.js +56 -0
- package/src/cli.js +182 -13
- package/src/fingerprint.js +10 -1
- package/src/index.js +302 -16
- package/src/input.js +4 -0
- package/src/mcp.js +7 -1
- package/src/metrics.js +49 -6
- package/src/ollama.js +37 -9
- package/src/platform/ios.js +30 -5
- package/src/refs.js +12 -1
- package/src/regions.js +54 -0
- package/src/screenmap.js +85 -6
- package/src/store.js +53 -0
- package/src/supervisor.js +20 -2
- package/src/view.js +84 -9
|
@@ -164,8 +164,21 @@ for (let round = 1; round <= rounds; round += 1) {
|
|
|
164
164
|
// means the navigation did not happen.
|
|
165
165
|
const s = fingerprint.similarity(previous.tokens, id.tokens ?? []);
|
|
166
166
|
if (s >= ARRIVAL_SUSPICION) {
|
|
167
|
+
// What was actually on screen, not just that it was the wrong thing.
|
|
168
|
+
//
|
|
169
|
+
// This check has fired twice on CI and both times the report was a
|
|
170
|
+
// similarity score and two screen names, which is enough to know the
|
|
171
|
+
// run is void and not enough to know why. The tour asserts arrival with
|
|
172
|
+
// a `waitFor` before this ever runs, so a failure here means the
|
|
173
|
+
// `waitFor` *passed* on a screen that was not the destination — and the
|
|
174
|
+
// labels are the only thing that can say what that screen was.
|
|
175
|
+
const labels = (id.entry?.targets ?? [])
|
|
176
|
+
.map((t) => t.label).filter(Boolean).slice(0, 12).map((l) => l.slice(0, 24));
|
|
167
177
|
arrivalFailures.push(
|
|
168
|
-
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
|
|
178
|
+
`${previous.name} -> ${screen.name}: similarity ${s.toFixed(2)} — the screen did not change`
|
|
179
|
+
+ `\n sensors: ${(id.entry?.sources ?? []).join('+') || 'none'}`
|
|
180
|
+
+ `, settled: ${id.settled}, frame ${Math.round(Date.now() - (id.state?.capturedAt ?? Date.now()))}ms old`
|
|
181
|
+
+ `\n on screen: ${labels.length ? labels.join(' · ') : '(nothing readable)'}`);
|
|
169
182
|
}
|
|
170
183
|
}
|
|
171
184
|
readings.push({
|
|
@@ -216,6 +229,12 @@ if (arrivalFailures.length) {
|
|
|
216
229
|
console.error('measure the tour rather than the fingerprint. Fix the tour and re-run.');
|
|
217
230
|
console.error('A `settle` step defaults to mode "stable", which returns instantly in the');
|
|
218
231
|
console.error('moment before an animation begins — action steps already settle on their own.');
|
|
232
|
+
console.error('');
|
|
233
|
+
console.error('The tour asserts arrival with a `waitFor` before any reading is taken, so a');
|
|
234
|
+
console.error('failure here means that wait PASSED on a screen that was not the destination.');
|
|
235
|
+
console.error('Read the labels above before changing the tour: either the wait matched');
|
|
236
|
+
console.error('something it should not have, or the reading came off a frame older than the');
|
|
237
|
+
console.error('navigation — and those have opposite fixes.');
|
|
219
238
|
process.exit(1);
|
|
220
239
|
}
|
|
221
240
|
|
|
@@ -76,6 +76,8 @@ if (baseline > 0.65) {
|
|
|
76
76
|
const STILL_MS_THRESHOLD = 3000;
|
|
77
77
|
console.log(`\nthe brief every model arm gets is ${ollama.readBrief().length} characters, read from native/supervise.swift`);
|
|
78
78
|
|
|
79
|
+
const withAbstain = process.argv.includes('--abstain');
|
|
80
|
+
|
|
79
81
|
const results = [];
|
|
80
82
|
for (const arm of arms) {
|
|
81
83
|
const judged = [];
|
|
@@ -87,6 +89,7 @@ for (const arm of arms) {
|
|
|
87
89
|
const detail = {};
|
|
88
90
|
const ruling = await supervisor.judge({
|
|
89
91
|
...r.situation,
|
|
92
|
+
mayAbstain: withAbstain,
|
|
90
93
|
options: { supervisor: arm },
|
|
91
94
|
// Wide on purpose. The shipped budget is 2.5s and a 14B will exceed it;
|
|
92
95
|
// capping here would score the larger model on *latency* while calling it
|
|
@@ -96,7 +99,7 @@ for (const arm of arms) {
|
|
|
96
99
|
detail,
|
|
97
100
|
});
|
|
98
101
|
if (!ruling) unanswered += 1;
|
|
99
|
-
judged.push({ r, ruling, detail });
|
|
102
|
+
judged.push({ r, ruling, detail, abstained: detail.kind === 'abstained' });
|
|
100
103
|
process.stdout.write('.');
|
|
101
104
|
}
|
|
102
105
|
const answered = judged.filter((j) => j.ruling);
|
|
@@ -104,6 +107,7 @@ for (const arm of arms) {
|
|
|
104
107
|
const lat = answered.map((j) => j.ruling.ms).filter(Number.isFinite).sort((a, b) => a - b);
|
|
105
108
|
results.push({
|
|
106
109
|
arm,
|
|
110
|
+
judged,
|
|
107
111
|
n: rows.length,
|
|
108
112
|
unanswered,
|
|
109
113
|
// Scored over every situation, not only the answered ones. A judge that
|
|
@@ -131,6 +135,24 @@ for (const r of results) {
|
|
|
131
135
|
console.log(`${r.arm.padEnd(26)} ${pct(r.accuracy).padStart(9)} ${`${r.medianMs ?? '—'}ms`.padStart(9)} ${`${r.unanswered}`.padStart(10)}`);
|
|
132
136
|
}
|
|
133
137
|
|
|
138
|
+
if (withAbstain) {
|
|
139
|
+
console.log(`\n--- item 100: the fourth word ---`);
|
|
140
|
+
console.log('The question is not whether it abstains. It is whether it abstains on');
|
|
141
|
+
console.log('the ones it would have got WRONG, or at random. A judge that declines');
|
|
142
|
+
console.log('uniformly has added latency and a round trip and bought nothing.\n');
|
|
143
|
+
console.log(`${'arm'.padEnd(22)} ${'answered'.padStart(9)} ${'of those'.padStart(9)} ${'abstained'.padStart(10)} ${'escalations'.padStart(12)}`);
|
|
144
|
+
console.log('-'.repeat(66));
|
|
145
|
+
for (const r of results) {
|
|
146
|
+
const answered = r.judged.filter((j) => j.ruling);
|
|
147
|
+
const right = answered.filter((j) => satisfies(j.ruling.decision, want(j.r))).length;
|
|
148
|
+
const abstained = r.judged.filter((j) => j.abstained).length;
|
|
149
|
+
console.log(`${r.arm.padEnd(22)} ${`${answered.length}/${rows.length}`.padStart(9)} ${pct(right / (answered.length || 1)).padStart(9)} ${`${abstained}`.padStart(10)} ${pct(abstained / rows.length).padStart(12)}`);
|
|
150
|
+
}
|
|
151
|
+
console.log('\n`of those` is accuracy on the questions it chose to answer. If the fourth');
|
|
152
|
+
console.log('word is working, that number is higher than the three-word accuracy above');
|
|
153
|
+
console.log('by more than the abstention rate would give by chance.');
|
|
154
|
+
}
|
|
155
|
+
|
|
134
156
|
console.log('\nerrors, by arm:');
|
|
135
157
|
for (const r of results) {
|
|
136
158
|
console.log(` ${r.arm}`);
|
|
@@ -140,6 +162,43 @@ for (const r of results) {
|
|
|
140
162
|
for (const [e, n] of [...seen].sort((a, b) => b[1] - a[1])) console.log(` ${String(n).padStart(3)}x ${e}`);
|
|
141
163
|
}
|
|
142
164
|
|
|
165
|
+
// --- the cascade: threshold -> local model -> Claude -----------------------
|
|
166
|
+
//
|
|
167
|
+
// The owner's proposal, and the numbers are the only way to say whether it
|
|
168
|
+
// helps: answer with the free rule where it is confident, fall through to the
|
|
169
|
+
// on-device model where it is not, and only then pay a round trip.
|
|
170
|
+
//
|
|
171
|
+
// **A cascade needs a tier that can decline, and neither tier has one.** The
|
|
172
|
+
// threshold is a comparison — it always answers. The supervisor's vocabulary is
|
|
173
|
+
// three words and none of them is "I don't know". So the interesting number is
|
|
174
|
+
// not "does a cascade help" but "how much would an abstain token be worth", and
|
|
175
|
+
// that is item 100. This measures it directly, by letting the rule abstain in a
|
|
176
|
+
// band around its own threshold and handing those to the next tier.
|
|
177
|
+
const armDecisions = new Map(results.map((r) => [r.arm, r.judged]));
|
|
178
|
+
console.log('\n--- cascade: the rule answers, the model covers where it abstains ---');
|
|
179
|
+
console.log(`${'abstain band'.padEnd(22)} ${'rule'.padStart(6)} ${'->model'.padStart(8)} ${'cascade'.padStart(9)} ${'model calls'.padStart(12)}`);
|
|
180
|
+
for (const band of [0, 250, 500, 750, 1000, 1500]) {
|
|
181
|
+
const lo = STILL_MS_THRESHOLD - band;
|
|
182
|
+
const hi = STILL_MS_THRESHOLD + band;
|
|
183
|
+
for (const arm of arms) {
|
|
184
|
+
const judged = armDecisions.get(arm) ?? [];
|
|
185
|
+
let right = 0;
|
|
186
|
+
let escalated = 0;
|
|
187
|
+
for (const [i, r] of rows.entries()) {
|
|
188
|
+
const still = r.situation.stillMs ?? 0;
|
|
189
|
+
if (band > 0 && still >= lo && still <= hi) {
|
|
190
|
+
escalated += 1;
|
|
191
|
+
const ruling = judged[i]?.ruling;
|
|
192
|
+
if (ruling && satisfies(ruling.decision, want(r))) right += 1;
|
|
193
|
+
continue;
|
|
194
|
+
}
|
|
195
|
+
if (satisfies(still > STILL_MS_THRESHOLD ? 'stop' : 'wait', want(r))) right += 1;
|
|
196
|
+
}
|
|
197
|
+
console.log(`${`+/-${band}ms -> ${arm}`.padEnd(22)} ${pct(stillRule).padStart(6)} ${`${escalated}`.padStart(8)} ${pct(right / rows.length).padStart(9)} ${`${escalated}/${rows.length}`.padStart(12)}`);
|
|
198
|
+
}
|
|
199
|
+
if (band === 0) console.log(' (band 0 = no abstention, the rule alone — every row below adds a tier)');
|
|
200
|
+
}
|
|
201
|
+
|
|
143
202
|
console.log('\nThis scores DECISIONS on identical inputs. It cannot score outcomes —');
|
|
144
203
|
console.log('whether acting on a ruling recovered the flow is a fact about the device at');
|
|
145
204
|
console.log('that moment, and belongs to whichever arm was live. See docs/EXPERIMENTS.md.');
|
package/src/actions.js
CHANGED
|
@@ -819,6 +819,22 @@ export async function runScript(
|
|
|
819
819
|
fingerprint: beforeScreen?.hash ?? null,
|
|
820
820
|
reason: 'verification_failed',
|
|
821
821
|
candidates: [],
|
|
822
|
+
// Assumed, not read. A verdict says the step did not do what was
|
|
823
|
+
// expected; it does not say which faculty would have prevented that.
|
|
824
|
+
//
|
|
825
|
+
// Measured in the field and it matters: on an app whose controls are
|
|
826
|
+
// largely unlabeled, `no-visible-change` came overwhelmingly from
|
|
827
|
+
// tapping an inert text label whose real hit target was an invisible
|
|
828
|
+
// chevron — icon semantics, Phase 15 — while this line filed every
|
|
829
|
+
// one as evidence about sense of time, Phase 11. The tester reached
|
|
830
|
+
// the right conclusion from their own notes and our instrument
|
|
831
|
+
// disagreed with them. It was wrong.
|
|
832
|
+
//
|
|
833
|
+
// Not reclassified here, because `no-visible-change` also covers a
|
|
834
|
+
// switch moving 0.1% of the screen, which is neither faculty. Saying
|
|
835
|
+
// "assumed" is the honest answer; guessing a better-sounding reason
|
|
836
|
+
// would be the same mistake in the other direction.
|
|
837
|
+
classified: false,
|
|
822
838
|
// `verification_failed` is the largest reason class in the log and it
|
|
823
839
|
// was the only one carrying no intent, which made most of the corpus
|
|
824
840
|
// useless for asking what kind of decision costs us. The step knows
|
|
@@ -854,6 +870,7 @@ export async function runScript(
|
|
|
854
870
|
// Carried from the throw site where it exists, and otherwise the step's
|
|
855
871
|
// own target — which is what was asked for either way.
|
|
856
872
|
intent: why.intent ?? goalOf(step),
|
|
873
|
+
classified: why.classified,
|
|
857
874
|
outcome: 'failed',
|
|
858
875
|
wallMs: Date.now() - stepStart,
|
|
859
876
|
detail: err.message,
|
|
@@ -2369,8 +2386,26 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2369
2386
|
timeoutMs: step.timeoutMs ?? 8000,
|
|
2370
2387
|
options: ctx.options,
|
|
2371
2388
|
});
|
|
2372
|
-
|
|
2373
|
-
|
|
2389
|
+
// Say where it is still moving — item 123.
|
|
2390
|
+
//
|
|
2391
|
+
// "did not settle within 25000ms" is true and unactionable, and a failed
|
|
2392
|
+
// step aborts the rest of the batch, so a spinner nobody cares about can
|
|
2393
|
+
// cost five steps that would have worked. The reporter who asked for this
|
|
2394
|
+
// had landed on `pause` plus `continueOnError` and called it *"strictly
|
|
2395
|
+
// worse than a settle that knows what to ignore"*. Knowing what to ignore
|
|
2396
|
+
// is the expensive half and is not built. Saying where is nearly free —
|
|
2397
|
+
// the signatures were already read to detect small changes — and it is
|
|
2398
|
+
// the difference between re-planning a flow and ignoring a corner of it.
|
|
2399
|
+
if (!w.satisfied) {
|
|
2400
|
+
if (w.stalled) throw new Error(w.live.note);
|
|
2401
|
+
throw new Error(`screen did not settle within ${w.waitedMs}ms — ${settleEvidence(w)}`);
|
|
2402
|
+
}
|
|
2403
|
+
// A settle that satisfied while part of the screen is still moving says
|
|
2404
|
+
// so. The daemon has boxed the moving part all along and nothing read it;
|
|
2405
|
+
// measured at 79s of claimed stillness on a screen animating at 85ms a
|
|
2406
|
+
// frame. Not treated as a failure — see the note on `animatingNow`.
|
|
2407
|
+
return `settled after ${w.waitedMs}ms`
|
|
2408
|
+
+ (w.animating ? ` (a ${w.animating.width}x${w.animating.height} region is still animating)` : '');
|
|
2374
2409
|
}
|
|
2375
2410
|
case 'waitText': {
|
|
2376
2411
|
const target = step.value ?? step.text;
|
|
@@ -2385,10 +2420,12 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2385
2420
|
// Same rule as `waitFor`, and `matchElement` says it in its own
|
|
2386
2421
|
// words: a query that matched several elements has found them all
|
|
2387
2422
|
// already.
|
|
2388
|
-
|
|
2389
|
-
|
|
2390
|
-
|
|
2391
|
-
|
|
2423
|
+
// Satisfies the wait, for the same reason as `waitFor` above: several
|
|
2424
|
+
// matches is an answer of "yes, it is here".
|
|
2425
|
+
const many = err.message.match(/matched (\d+) elements/);
|
|
2426
|
+
if (many) {
|
|
2427
|
+
return `"${target}" is on screen (${many[1]} matches)`
|
|
2428
|
+
+ ' — the wait is satisfied; pass an index to act on one of them';
|
|
2392
2429
|
}
|
|
2393
2430
|
}
|
|
2394
2431
|
await sleep(250);
|
|
@@ -2553,10 +2590,22 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2553
2590
|
//
|
|
2554
2591
|
// Distinguished by the tag at the throw site rather than by reading
|
|
2555
2592
|
// the message, because "not on this screen" tags the same reason.
|
|
2593
|
+
//
|
|
2594
|
+
// And it **satisfies** the wait rather than failing it, which is the
|
|
2595
|
+
// correction. The paragraph above had the reasoning right — ambiguous
|
|
2596
|
+
// means the target is present, several times over — and then threw
|
|
2597
|
+
// anyway. A `waitFor` asks one question, *has it arrived*, and two
|
|
2598
|
+
// matches is a yes. Reported from the field: the batch was abandoned
|
|
2599
|
+
// and three queued steps discarded on a screen that was exactly where
|
|
2600
|
+
// the flow wanted to be, and the tester had to re-issue the lot with
|
|
2601
|
+
// an index. The ambiguity is real and belongs in the next step's
|
|
2602
|
+
// selector, not in this step's verdict.
|
|
2556
2603
|
if (metrics.escalationOf(err)?.ambiguous) {
|
|
2557
|
-
|
|
2558
|
-
|
|
2559
|
-
);
|
|
2604
|
+
// The count comes from the message because the tag is a boolean —
|
|
2605
|
+
// it marks *which kind* of ambiguity, not how many.
|
|
2606
|
+
const n = err.message.match(/matches (\d+) things/)?.[1];
|
|
2607
|
+
return `${query} is on screen${n ? ` (${n} matches)` : ' more than once'}`
|
|
2608
|
+
+ ' — the wait is satisfied; pass an index or a #ref to act on one of them';
|
|
2560
2609
|
}
|
|
2561
2610
|
}
|
|
2562
2611
|
if (Date.now() >= limit) break;
|
|
@@ -2657,3 +2706,41 @@ async function runStep(deviceQuery, udid, step, ctx) {
|
|
|
2657
2706
|
throw new Error(`unknown step "${step.action}"`);
|
|
2658
2707
|
}
|
|
2659
2708
|
}
|
|
2709
|
+
|
|
2710
|
+
/**
|
|
2711
|
+
* Why a settle gave up, in the numbers that decided it.
|
|
2712
|
+
*
|
|
2713
|
+
* `screen did not settle within 25008ms` names the one quantity that is never
|
|
2714
|
+
* the reason, and two CI cycles went into guessing what was behind it. A wait
|
|
2715
|
+
* turns on three things and the message carried none of them:
|
|
2716
|
+
*
|
|
2717
|
+
* - how long the screen had actually been still when we gave up
|
|
2718
|
+
* - how much stillness was being asked for
|
|
2719
|
+
* - whether any frame arrived during the call at all
|
|
2720
|
+
*
|
|
2721
|
+
* *"Still for 1,200 ms of the 1,400 required"* and *"still for 135,000 ms and no
|
|
2722
|
+
* frame ever arrived"* are opposite diagnoses — the first is a budget a hair too
|
|
2723
|
+
* tight, the second is a stalled pipeline wearing a calm screen — and they had
|
|
2724
|
+
* one sentence between them. So did *"something is animating in the top right"*.
|
|
2725
|
+
*
|
|
2726
|
+
* Ordered by what to do next: what is moving, then how close stillness got,
|
|
2727
|
+
* then whether anything was observed, then black frames.
|
|
2728
|
+
*/
|
|
2729
|
+
export function settleEvidence(w) {
|
|
2730
|
+
const parts = [];
|
|
2731
|
+
const m = w.motion;
|
|
2732
|
+
if (m) parts.push(`the movement is ${m.where}${m.localised ? ` (${m.share}% of it)` : ''}`);
|
|
2733
|
+
if (Number.isFinite(w.stillForMs) && Number.isFinite(w.stableMsRequired)) {
|
|
2734
|
+
parts.push(`still for ${w.stillForMs}ms of the ${w.stableMsRequired}ms required`);
|
|
2735
|
+
}
|
|
2736
|
+
// Zero frames is the loud one: a wait that observed nothing has not measured
|
|
2737
|
+
// this screen, it has measured a file on disk.
|
|
2738
|
+
if (Number.isFinite(w.framesSeen)) {
|
|
2739
|
+
parts.push(w.framesSeen > 0
|
|
2740
|
+
? `${w.framesSeen} frame(s) arrived while waiting`
|
|
2741
|
+
: 'NO frame arrived while waiting — capture, not the screen');
|
|
2742
|
+
}
|
|
2743
|
+
if (w.blackFrames > 0) parts.push(`${w.blackFrames} black frame(s) — see the capture wedge`);
|
|
2744
|
+
return (parts.length ? parts.join('; ') : 'nothing observed')
|
|
2745
|
+
+ (m ? `\n${m.map}` : '');
|
|
2746
|
+
}
|
package/src/analyze.js
CHANGED
|
@@ -65,6 +65,25 @@ export function signatureDiff(a, b) {
|
|
|
65
65
|
*/
|
|
66
66
|
export const CELL_CHANGE = 0.012;
|
|
67
67
|
|
|
68
|
+
/**
|
|
69
|
+
* How far two *capture paths* may differ and still be looking at one screen.
|
|
70
|
+
*
|
|
71
|
+
* Not the change threshold, and the distinction matters. `signatureDiff > 0.004`
|
|
72
|
+
* asks whether a screen moved between two frames from the *same* path. This asks
|
|
73
|
+
* whether the daemon's frame and an independent `simctl` screenshot show the
|
|
74
|
+
* same thing — and they never match closely, because one is downscaled by the
|
|
75
|
+
* capture loop and the other is a full-resolution PNG scaled here.
|
|
76
|
+
*
|
|
77
|
+
* Measured on this device, which is the only reason a number appears:
|
|
78
|
+
*
|
|
79
|
+
* same screen, two paths 0.00123 (three runs, identical to five places)
|
|
80
|
+
* two different screens 0.68603 (before and after a home press)
|
|
81
|
+
*
|
|
82
|
+
* A separation of 550x, so the threshold is not delicate. 0.02 is sixteen times
|
|
83
|
+
* the scaling cost and thirty-four times under the signal.
|
|
84
|
+
*/
|
|
85
|
+
export const PATHS_AGREE = 0.02;
|
|
86
|
+
|
|
68
87
|
/** The largest single-region change between two signatures, 0-1. */
|
|
69
88
|
export function maxCellDelta(a, b) {
|
|
70
89
|
const deltas = regionDeltas(a, b);
|
|
@@ -100,6 +119,43 @@ export function regionMap(deltas, cols = REGION_COLS) {
|
|
|
100
119
|
return lines.join('\n');
|
|
101
120
|
}
|
|
102
121
|
|
|
122
|
+
/**
|
|
123
|
+
* Where the movement is, in words — item 123.
|
|
124
|
+
*
|
|
125
|
+
* A settle that gives up says only that it gave up, and the reporter who asked
|
|
126
|
+
* for this put the cost plainly: a streaming summary panel, a Lottie and a
|
|
127
|
+
* support widget never settle, so the settle fails, and because a failed step
|
|
128
|
+
* aborts the batch it takes the other five steps with it. Their workaround was
|
|
129
|
+
* `pause` plus `continueOnError`, which they called *"strictly worse than a
|
|
130
|
+
* settle that knows what to ignore"*. Knowing what to ignore is the expensive
|
|
131
|
+
* half. Saying **where** is nearly free, and it is what turns "did not settle"
|
|
132
|
+
* into "there is a spinner in the top-right and nobody cares about it".
|
|
133
|
+
*
|
|
134
|
+
* Deliberately coarse. Thirds of the screen in each axis, named the way a person
|
|
135
|
+
* would point at it, and a share so a caller can tell one spinner from a screen
|
|
136
|
+
* that is genuinely still in flight.
|
|
137
|
+
*/
|
|
138
|
+
export function describeMotion(deltas, cols = REGION_COLS) {
|
|
139
|
+
const total = deltas.reduce((a, b) => a + b, 0);
|
|
140
|
+
if (!deltas.length || total <= 0) return null;
|
|
141
|
+
const rows = Math.ceil(deltas.length / cols);
|
|
142
|
+
const bands = new Map();
|
|
143
|
+
const third = (i, n) => (i < n / 3 ? 0 : i < (2 * n) / 3 ? 1 : 2);
|
|
144
|
+
const DOWN = ['top', 'middle', 'bottom'];
|
|
145
|
+
const ACROSS = ['left', 'centre', 'right'];
|
|
146
|
+
deltas.forEach((d, i) => {
|
|
147
|
+
const key = `${DOWN[third(Math.floor(i / cols), rows)]}-${ACROSS[third(i % cols, cols)]}`;
|
|
148
|
+
bands.set(key, (bands.get(key) ?? 0) + d);
|
|
149
|
+
});
|
|
150
|
+
const ranked = [...bands.entries()].sort((a, b) => b[1] - a[1]);
|
|
151
|
+
const [where, amount] = ranked[0];
|
|
152
|
+
const share = Math.round((amount / total) * 100);
|
|
153
|
+
// A single band holding most of the movement is a localised animation; spread
|
|
154
|
+
// evenly, the whole screen is in flight and naming a corner would mislead.
|
|
155
|
+
if (share < 40) return { where: 'spread across the screen', share, localised: false };
|
|
156
|
+
return { where: where.replace('-', ' '), share, localised: true };
|
|
157
|
+
}
|
|
158
|
+
|
|
103
159
|
/** Signatures live in state.json, so they are stored as compact hex. */
|
|
104
160
|
export function signatureToHex(sig) {
|
|
105
161
|
return sig.map((v) => v.toString(16).padStart(2, '0')).join('');
|
package/src/cli.js
CHANGED
|
@@ -467,6 +467,77 @@ async function main() {
|
|
|
467
467
|
}
|
|
468
468
|
|
|
469
469
|
case 'frame': {
|
|
470
|
+
// `--fresh` captures independently of the daemon, and then says whether
|
|
471
|
+
// the two agree.
|
|
472
|
+
//
|
|
473
|
+
// This is the arbiter three field reports had to leave simframe to get.
|
|
474
|
+
// When `sim_look` served a three-hour-stale image labelled `130ms old`,
|
|
475
|
+
// the thing that finally settled it was `xcrun simctl io … screenshot` —
|
|
476
|
+
// run by hand, outside the tool, because nothing inside offered an
|
|
477
|
+
// independent read. Worse, the obvious candidate lies: `--engine` decides
|
|
478
|
+
// how to *start* a daemon, so passing `--engine=screenshot` to a read
|
|
479
|
+
// command returns the running daemon's cached frame. A tester compared
|
|
480
|
+
// the two, got byte-identical files with the same frame number, and
|
|
481
|
+
// reasonably concluded "the fallback engine is not an escape hatch".
|
|
482
|
+
//
|
|
483
|
+
// One command now answers the question the escape hatch was for: capture
|
|
484
|
+
// the screen twice by two different paths and report whether they agree.
|
|
485
|
+
if (flags.fresh) {
|
|
486
|
+
const dev = await resolveDevice(device);
|
|
487
|
+
const out = flags.out || path.join(process.cwd(), 'simframe.png');
|
|
488
|
+
await screenshot(dev.udid, out);
|
|
489
|
+
const png = fs.readFileSync(out);
|
|
490
|
+
// The daemon's own newest frame, for comparison. Absent is fine and
|
|
491
|
+
// interesting in itself: an independent capture that works while the
|
|
492
|
+
// daemon has none is exactly the wedge.
|
|
493
|
+
let cached = null;
|
|
494
|
+
try {
|
|
495
|
+
cached = await api.getFrame(device, { detail: flags.detail ?? 'normal', options });
|
|
496
|
+
} catch { /* no daemon, or it has nothing — reported below */ }
|
|
497
|
+
// Compared by *content*, never by bytes. The direct capture is a
|
|
498
|
+
// full-resolution PNG and the daemon's is downscaled, so a byte
|
|
499
|
+
// comparison says "different" every time — which is a confident wrong
|
|
500
|
+
// answer about the one question this command exists to settle. Both
|
|
501
|
+
// are decoded and reduced to the same region signature the change
|
|
502
|
+
// detector already uses, and `signatureDiff` is the same measure that
|
|
503
|
+
// decides whether a screen moved.
|
|
504
|
+
let diff = null;
|
|
505
|
+
if (cached) {
|
|
506
|
+
try {
|
|
507
|
+
const a = analyze.regionSignature(decodePng(png));
|
|
508
|
+
const b = analyze.regionSignature(decodePng(cached.png));
|
|
509
|
+
diff = analyze.signatureDiff(a, b);
|
|
510
|
+
} catch { /* an undecodable frame is reported as "could not compare" */ }
|
|
511
|
+
}
|
|
512
|
+
// The threshold the daemon itself calls a change. Below it the two
|
|
513
|
+
// paths are looking at the same screen.
|
|
514
|
+
const same = diff == null ? null : diff <= analyze.PATHS_AGREE;
|
|
515
|
+
emit(
|
|
516
|
+
flags,
|
|
517
|
+
{
|
|
518
|
+
file: out,
|
|
519
|
+
fresh: true,
|
|
520
|
+
bytes: png.length,
|
|
521
|
+
daemonSeq: cached?.state?.seq ?? null,
|
|
522
|
+
daemonAgeMs: cached?.ageMs ?? null,
|
|
523
|
+
agrees: same,
|
|
524
|
+
difference: diff == null ? null : Number(diff.toFixed(4)),
|
|
525
|
+
},
|
|
526
|
+
[
|
|
527
|
+
`${out} — captured directly from the device, ${png.length} bytes`,
|
|
528
|
+
cached
|
|
529
|
+
? `the daemon's newest frame is #${cached.state.seq}, ${cached.ageMs}ms old`
|
|
530
|
+
+ (same == null
|
|
531
|
+
? ' — could not be compared (one of the two would not decode)'
|
|
532
|
+
: same
|
|
533
|
+
? ` — the two paths agree (difference ${diff.toFixed(4)}, under the ${analyze.PATHS_AGREE} two paths may differ by)`
|
|
534
|
+
: ` — they DISAGREE (difference ${diff.toFixed(4)}). Two capture paths see different screens;`
|
|
535
|
+
+ ' this file is the one that bypassed the daemon. `simframe revive` re-attaches capture.')
|
|
536
|
+
: 'the daemon has no frame to compare against, while a direct capture worked',
|
|
537
|
+
],
|
|
538
|
+
);
|
|
539
|
+
return;
|
|
540
|
+
}
|
|
470
541
|
const res = await api.getFrame(device, { detail: flags.detail ?? 'normal', options });
|
|
471
542
|
const out = flags.out || path.join(process.cwd(), 'simframe.png');
|
|
472
543
|
fs.writeFileSync(out, res.png);
|
|
@@ -491,7 +562,9 @@ async function main() {
|
|
|
491
562
|
} else {
|
|
492
563
|
const s = res.state;
|
|
493
564
|
const out = [];
|
|
494
|
-
|
|
565
|
+
// Any note, not only a failing one: a dead surface reports `ok` with
|
|
566
|
+
// something important to say. See `liveness`.
|
|
567
|
+
if (res.live.note) out.push(`WARNING: ${res.live.note}`);
|
|
495
568
|
// A cause, rather than five silent no-ops. Every tap on a stale
|
|
496
569
|
// session is dispatched successfully and moves nothing.
|
|
497
570
|
if (res.input?.stale) out.push(`input: stale — ${res.input.reason}`);
|
|
@@ -546,20 +619,36 @@ async function main() {
|
|
|
546
619
|
changedBeforeWait: Boolean(res.changedBeforeWait),
|
|
547
620
|
noVisibleChange: Boolean(res.noVisibleChange),
|
|
548
621
|
stalled: Boolean(res.stalled),
|
|
622
|
+
// Where it was still moving, when it never stopped — item 123.
|
|
623
|
+
//
|
|
624
|
+
// `waitFor` has computed this since it learned to, and this payload
|
|
625
|
+
// is hand-built, so the field existed and no caller could see it. I
|
|
626
|
+
// read a null here and nearly concluded the tracking was broken; it
|
|
627
|
+
// was the reporting.
|
|
628
|
+
motion: res.motion ?? null,
|
|
629
|
+
animating: res.animating ?? null,
|
|
549
630
|
hash: res.state?.hash,
|
|
550
631
|
seq: res.state?.seq,
|
|
551
632
|
},
|
|
552
633
|
() => {
|
|
553
634
|
if (res.satisfied) {
|
|
554
635
|
return `${res.mode === 'change' ? 'changed' : 'settled'} after ${res.waitedMs}ms — frame #${res.state.seq}` +
|
|
555
|
-
(res.changedBeforeWait ? ' (change had already happened before the call)' : '')
|
|
636
|
+
(res.changedBeforeWait ? ' (change had already happened before the call)' : '') +
|
|
637
|
+
(res.animating
|
|
638
|
+
? `\nbut a ${res.animating.width}x${res.animating.height} region is still animating`
|
|
639
|
+
+ ' — the stillness signal is a mean and cannot see it'
|
|
640
|
+
: '');
|
|
556
641
|
}
|
|
557
642
|
if (res.noVisibleChange) {
|
|
558
643
|
return `no visible change after ${res.waitedMs}ms — screen stable, nothing moved (the action may have had no visible effect)`;
|
|
559
644
|
}
|
|
560
645
|
if (res.stalled) return `capture stalled after ${res.waitedMs}ms — ${res.live.note}`;
|
|
561
646
|
return `timed out after ${res.waitedMs}ms — no ${res.mode === 'change' ? 'change' : 'settle'}` +
|
|
562
|
-
(res.sawChange ? '' : '; if the change happened before this call, pass `--since` from `simframe mark`')
|
|
647
|
+
(res.sawChange ? '' : '; if the change happened before this call, pass `--since` from `simframe mark`') +
|
|
648
|
+
(res.motion
|
|
649
|
+
? `\nthe movement is ${res.motion.where}`
|
|
650
|
+
+ `${res.motion.localised ? ` (${res.motion.share}% of it)` : ''}:\n${res.motion.map}`
|
|
651
|
+
: '');
|
|
563
652
|
},
|
|
564
653
|
);
|
|
565
654
|
process.exitCode = res.satisfied ? 0 : 1;
|
|
@@ -1106,13 +1195,31 @@ async function main() {
|
|
|
1106
1195
|
|
|
1107
1196
|
case 'supervisions': {
|
|
1108
1197
|
const dev = await resolveDevice(flags.device);
|
|
1109
|
-
|
|
1198
|
+
// `--session` works here now, and did not before.
|
|
1199
|
+
//
|
|
1200
|
+
// Reported twice from the field: two different session ids returned
|
|
1201
|
+
// byte-identical output while `escalations --session` filtered correctly.
|
|
1202
|
+
// A flag that exists on one command and is silently inert on its sibling
|
|
1203
|
+
// is worse than an absent one — this command's own footer warns that the
|
|
1204
|
+
// counts pool multiple agents and then offered no way to unpool them.
|
|
1205
|
+
const session = flags.session === true
|
|
1206
|
+
? metrics.sessionId()
|
|
1207
|
+
: (flags.session ? String(flags.session) : null);
|
|
1208
|
+
const all = metrics.readSupervisions(dev.udid, { limit: flags.last ? num(flags.last) : undefined });
|
|
1209
|
+
const records = session ? all.filter((r) => r?.session_id === session) : all;
|
|
1110
1210
|
const b = metrics.supervisionBreakdown(records);
|
|
1111
1211
|
if (flags.out) store.writeAtomic(String(flags.out), `${JSON.stringify({ ...b, records }, null, 2)}\n`);
|
|
1112
1212
|
emit(flags, { ...b, records: flags.verbose ? records : undefined }, [
|
|
1113
1213
|
`${b.total} supervisor ruling${b.total === 1 ? '' : 's'} on ${dev.name}`,
|
|
1114
|
-
|
|
1115
|
-
|
|
1214
|
+
// "Nothing here" and "nothing matched your filter" are different
|
|
1215
|
+
// answers, and the first one told a reader the supervisor had never
|
|
1216
|
+
// run on a device holding 111 rulings.
|
|
1217
|
+
b.total
|
|
1218
|
+
? null
|
|
1219
|
+
: (all.length
|
|
1220
|
+
? `no ruling in this log belongs to session ${session} — the device has ${all.length}.`
|
|
1221
|
+
: 'Nothing has been judged on this device yet. The supervisor is off unless'
|
|
1222
|
+
+ ' SIMFRAME_SUPERVISOR=apple, and a ruling is only recorded when a step actually fails.'),
|
|
1116
1223
|
...Object.entries(b.decision_to_outcome)
|
|
1117
1224
|
.sort((a, c) => c[1] - a[1])
|
|
1118
1225
|
.map(([k, n]) => ` ${k.padEnd(28)} ${String(n).padStart(4)}`),
|
|
@@ -1128,8 +1235,12 @@ async function main() {
|
|
|
1128
1235
|
? ` — ${b.p95_unknown} ruling(s) are on edges with no p95, so they cannot take part in 101's comparison`
|
|
1129
1236
|
: '')
|
|
1130
1237
|
: null,
|
|
1238
|
+
session && all.length !== records.length
|
|
1239
|
+
? `filtered to session ${session}: ${records.length} of ${all.length} ruling(s)`
|
|
1240
|
+
: null,
|
|
1131
1241
|
b.sessions.length > 1
|
|
1132
1242
|
? `WARNING ${b.sessions.length} sessions are pooled here; two agents on one device write one file`
|
|
1243
|
+
+ ' — narrow with --session (this process) or --session=<id>'
|
|
1133
1244
|
: null,
|
|
1134
1245
|
].filter((l) => l !== null).join('\n'));
|
|
1135
1246
|
break;
|
|
@@ -1148,10 +1259,29 @@ async function main() {
|
|
|
1148
1259
|
...metrics.REASONS
|
|
1149
1260
|
.filter((r) => b.by_reason[r])
|
|
1150
1261
|
.sort((a, c) => b.by_reason[c] - b.by_reason[a])
|
|
1151
|
-
.map((r) =>
|
|
1152
|
-
|
|
1262
|
+
.map((r) => {
|
|
1263
|
+
const n = b.by_reason[r];
|
|
1264
|
+
const read = b.classified_by_reason?.[r] ?? 0;
|
|
1265
|
+
// A faculty is only named for the part of a reason that was read
|
|
1266
|
+
// off the failure. The rest is a count of things nothing could
|
|
1267
|
+
// classify, and naming a phase against it is advice with nothing
|
|
1268
|
+
// behind it — which is how this report came to tell a tester that
|
|
1269
|
+
// their unlabeled-control problem was a timing problem.
|
|
1270
|
+
const assumed = b.assumed_by_reason?.[r] ?? 0;
|
|
1271
|
+
const named = metrics.BUILT_FACULTIES.has(metrics.FACULTY[r])
|
|
1153
1272
|
? `not removed by: ${metrics.FACULTY[r]} [built]`
|
|
1154
|
-
: `would be removed by: ${metrics.FACULTY[r]}
|
|
1273
|
+
: `would be removed by: ${metrics.FACULTY[r]}`;
|
|
1274
|
+
let verdict;
|
|
1275
|
+
if (read > 0) {
|
|
1276
|
+
verdict = named + (read < n ? ` (on the ${read} of ${n} whose reason was read)` : '');
|
|
1277
|
+
} else if (assumed > 0) {
|
|
1278
|
+
verdict = 'reason assumed, not read — no faculty can be named from these';
|
|
1279
|
+
} else {
|
|
1280
|
+
// Neither read nor assumed: the log predates the distinction.
|
|
1281
|
+
verdict = `${named} — but these records predate the check, so treat it as untested`;
|
|
1282
|
+
}
|
|
1283
|
+
return ` ${r.padEnd(20)} ${String(n).padStart(4)} ${verdict}`;
|
|
1284
|
+
}),
|
|
1155
1285
|
b.total ? '' : null,
|
|
1156
1286
|
b.total ? `avoidable ${b.avoidable}/${b.total} (${b.avoidable_escalation_rate})` : null,
|
|
1157
1287
|
// Said out loud rather than left for someone to discover: the rate is
|
|
@@ -1525,9 +1655,38 @@ async function doctor({ json = false, strict = false, device, options = {} } = {
|
|
|
1525
1655
|
// machine could be doing better and silently is not; a driver someone
|
|
1526
1656
|
// selected on purpose is neither silent nor a surprise.
|
|
1527
1657
|
const axState = !ax.available ? 'optional' : ax.name === 'simframed' || ax.chosen ? 'ok' : 'warn';
|
|
1528
|
-
|
|
1529
|
-
|
|
1658
|
+
// Prove a round trip, not a presence — the same correction this file
|
|
1659
|
+
// already made for the supervisor, never carried across to here.
|
|
1660
|
+
//
|
|
1661
|
+
// A CI run read the screen eighteen times and every single reading came
|
|
1662
|
+
// back `ocr` with no `ax` at all, while this check said `ok` because a
|
|
1663
|
+
// driver was configured. It is configured; it answers with nothing. Six
|
|
1664
|
+
// minutes later the fingerprint step failed with a distribution mystery,
|
|
1665
|
+
// and the layer that had actually died was named nowhere. Asking the tree
|
|
1666
|
+
// for the current screen costs one read (~50ms) and turns that into a
|
|
1667
|
+
// first-minute failure with the right sentence on it.
|
|
1668
|
+
let axCount = null;
|
|
1669
|
+
if (ax.available) {
|
|
1670
|
+
try {
|
|
1671
|
+
axCount = (await input.describeAll(d.udid)).length;
|
|
1672
|
+
} catch {
|
|
1673
|
+
axCount = 0;
|
|
1674
|
+
}
|
|
1675
|
+
}
|
|
1676
|
+
add(`accessibility tree (${d.name})`,
|
|
1677
|
+
// `warn`, not `fail`: a genuinely empty screen exists — a springboard
|
|
1678
|
+
// mid-boot, a black frame — and a hard error on one would cry wolf.
|
|
1679
|
+
// The count is exported so a caller that knows the screen is not empty
|
|
1680
|
+
// can assert on it, which is what CI does.
|
|
1681
|
+
ax.available && axCount === 0 ? 'warn' : axState,
|
|
1682
|
+
ax.available
|
|
1683
|
+
? `${ax.name}: ${ax.version}`
|
|
1684
|
+
+ (axCount === 0
|
|
1685
|
+
? ' — but it returned NO elements for the current screen, so every read is OCR alone'
|
|
1686
|
+
: axCount != null ? `; ${axCount} element(s) on the current screen` : '')
|
|
1687
|
+
: `unavailable: ${ax.reason}`,
|
|
1530
1688
|
{ key: 'ax.driver', value: ax.name });
|
|
1689
|
+
if (axCount != null) add(null, null, null, { key: 'ax.elements', value: axCount });
|
|
1531
1690
|
}
|
|
1532
1691
|
if (probed.length) {
|
|
1533
1692
|
const t0 = Date.now();
|
|
@@ -1554,6 +1713,13 @@ async function doctor({ json = false, strict = false, device, options = {} } = {
|
|
|
1554
1713
|
for (const d of probed) {
|
|
1555
1714
|
const live = api.liveness(d.udid, (await api.getState(d.udid)).state);
|
|
1556
1715
|
if (live.stalled) add(`capture health (${d.name})`, 'fail', live.note, { key: 'capture.stalled', value: true });
|
|
1716
|
+
// A `warn` rather than a `fail`, because a genuinely inert screen is
|
|
1717
|
+
// possible and this is a contradiction between two numbers rather than
|
|
1718
|
+
// a proven fault. It is still the loudest thing `doctor` can say about
|
|
1719
|
+
// the failure that made a tester report a false application state.
|
|
1720
|
+
else if (live.suspectSurface) {
|
|
1721
|
+
add(`capture surface (${d.name})`, 'warn', live.note, { key: 'capture.suspectSurface', value: true });
|
|
1722
|
+
}
|
|
1557
1723
|
}
|
|
1558
1724
|
}
|
|
1559
1725
|
} catch (err) {
|
|
@@ -1594,11 +1760,14 @@ async function doctor({ json = false, strict = false, device, options = {} } = {
|
|
|
1594
1760
|
warnings: warned.length,
|
|
1595
1761
|
optional: optional.length,
|
|
1596
1762
|
...flat,
|
|
1597
|
-
checks: checks.map(({ name, level, detail }) => ({ name, level, detail })),
|
|
1763
|
+
checks: checks.filter((c) => c.name).map(({ name, level, detail }) => ({ name, level, detail })),
|
|
1598
1764
|
}, null, 2));
|
|
1599
1765
|
} else {
|
|
1600
1766
|
const mark = { ok: 'ok ', warn: 'WARN', fail: 'FAIL', optional: '-- ' };
|
|
1601
|
-
for
|
|
1767
|
+
// A nameless entry is data for `--json` and not a line for a reader — the
|
|
1768
|
+
// element count belongs beside the layer it describes, not on a row of its
|
|
1769
|
+
// own.
|
|
1770
|
+
for (const c of checks) if (c.name) console.log(`${mark[c.level]} ${c.name.padEnd(24)} ${c.detail}`);
|
|
1602
1771
|
if (warned.length) {
|
|
1603
1772
|
console.log(`\n${warned.length} layer(s) degraded. simframe still works, but not at full speed or coverage:`);
|
|
1604
1773
|
for (const c of warned) console.log(` - ${c.name}: ${c.detail}`);
|
package/src/fingerprint.js
CHANGED
|
@@ -72,8 +72,17 @@ import * as regions from './regions.js';
|
|
|
72
72
|
* graph merged them. Both bounds are absolute now. Screens that were
|
|
73
73
|
* missed at 7 hash differently at 8, and unlike a stale hash that matches
|
|
74
74
|
* nothing, these matched the *wrong* thing.
|
|
75
|
+
*
|
|
76
|
+
* 9 — nothing in this file changed. The element list it is given did: item 122
|
|
77
|
+
* stopped dropping accessibility nodes that have no name, so a screen with
|
|
78
|
+
* an icon-only control now carries a token for it that it did not carry
|
|
79
|
+
* before. That is a better identity — a nav bar with an overflow menu and
|
|
80
|
+
* one without are not the same screen — and it is still a different hash
|
|
81
|
+
* for the same screen, which is what this number exists to declare. The
|
|
82
|
+
* lesson worth keeping is that the rules version is not a version of *this
|
|
83
|
+
* file*; it is a version of the token set, and the token set has an input.
|
|
75
84
|
*/
|
|
76
|
-
export const TOKEN_RULES_VERSION =
|
|
85
|
+
export const TOKEN_RULES_VERSION = 9;
|
|
77
86
|
|
|
78
87
|
/** Frames are quantised to this, so sub-pixel drift and a nudged row do not matter. */
|
|
79
88
|
export const GRID = 24;
|